{"data":{"slug":"alibaba-rtp-llm","name":"rtp-llm","tagline":"Alibaba's high-performance LLM inference engine for diverse applications.","github_url":"https://github.com/alibaba/rtp-llm","owner":"alibaba","repo":"rtp-llm","owner_avatar_url":"https://avatars.githubusercontent.com/u/1961952?v=4","primary_language":"Cuda","stars":1312,"forks":260,"topics":["gpt","inference","llama","llm","llm-serving","llmops","model-serving"],"archived":false,"github_pushed_at":"2026-08-20T17:20:55+00:00","maintenance_label":"Very active","stars_delta_30d":30,"url":"https://www.graphcanon.com/tools/alibaba-rtp-llm","markdown_url":"https://www.graphcanon.com/tools/alibaba-rtp-llm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alibaba-rtp-llm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alibaba-rtp-llm","description":"RTP-LLM: Alibaba's high-performance LLM inference engine for diverse applications.","homepage_url":null,"license":"Apache-2.0","open_issues":194,"watchers":19,"ai_summary":"RTP-LLM is an inference engine that provides high performance for various large language model (LLM) applications.","readme_excerpt":"## Getting Started\n- [Install RTP-LLM](https://rtp-llm.ai/build/en/start/install.html)\n- [Quick Start](https://rtp-llm.ai/build/en/backend/send_request.html)\n- [Backend Tutorial](https://rtp-llm.ai/build/en/references/deepseek/index.html)\n- [Contribution Guide](https://rtp-llm.ai/build/en/references/Contributing.html)","github_created_at":"2023-12-27T08:22:59+00:00","created_at":"2026-07-07T17:42:56.109855+00:00","updated_at":"2026-08-20T18:01:41.22019+00:00","categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"gpt","name":"gpt"},{"slug":"inference","name":"inference"},{"slug":"llama","name":"llama"},{"slug":"llm","name":"llm"},{"slug":"llm-serving","name":"llm-serving"},{"slug":"llmops","name":"llmops"},{"slug":"model-serving","name":"model-serving"}],"trust":{"provenance":{"is_fork":false,"github_id":736191625,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-08-20T18:01:40.479Z","maintenance":{"label":"Very active","score":96,"methodology":"github_public_v1","releases_90d":0,"days_since_push":0,"last_release_at":"2025-10-31T07:54:08Z","stars_delta_30d":30,"open_issues_delta_30d":32},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-11T11:22:21.386Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-08-20T18:01:40.938Z"},"languages":{"value":["cuda"],"source":"github.language","observed_at":"2026-08-20T18:01:40.938Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-08-20T18:01:40.938Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":{"notes":["Requires CUDA configuration and NVIDIA GPU availability to exploit its full performance capabilities."]},"constraints":null,"when_to_use":["When you are looking for a tool that leverages CUDA-based optimization for deploying Large Language Models (LLMs) across diverse applications.","If your project requires high-performance inference capabilities, particularly when running on NVIDIA GPUs due to its CUDA foundation, RTP-LLM can provide significant efficiency.","For seamless integration with Alibaba's ecosystem or projects that wish to leverage Alibaba’s specific optimizations."],"when_not_to_use":["When your development environment does not support CUDA as RTP-LLM is primarily based on it and might perform inadequately without direct GPU-acceleration from NVIDIA.","If your project has strict licensing constraints; while the Apache-2.0 license is permissive, certain projects may require tools with different or more restrictive licenses to meet compliance needs.","For applications that do not align well with Alibaba’s backend setup and prefer a more independent or competitor-based solution for their LLM serving needs."],"source":"enrich:decision_facts","observed_at":"2026-07-11T03:29:00.262Z"},"constraint_facets":null,"decision_summary":[{"label":"Requirements","value":"Requires CUDA configuration and NVIDIA GPU availability to exploit its full performance capabilities."},{"label":"Adopt for","value":"RTP-LLM is Alibaba's high-performance inference engine for LLMs, specifically designed and optimized with CUDA. It supports a variety of applications from GPT to LLaMA models."},{"label":"License detail","value":"The tool operates under the Apache-2.0 license making it freely available for commercial use, provided proper attribution is given."}]}}