{"data":{"slug":"microsoft-sarathi-serve","name":"sarathi-serve","tagline":"A low-latency and high-throughput serving engine for LLMs","github_url":"https://github.com/microsoft/sarathi-serve","owner":"microsoft","repo":"sarathi-serve","owner_avatar_url":"https://avatars.githubusercontent.com/u/6154722?v=4","primary_language":"Python","stars":520,"forks":65,"topics":["llama","llm-inference","pytorch","transformer"],"archived":false,"github_pushed_at":"2026-01-08T05:10:57+00:00","maintenance_label":"Slowing","stars_delta_30d":8,"url":"https://www.graphcanon.com/tools/microsoft-sarathi-serve","markdown_url":"https://www.graphcanon.com/tools/microsoft-sarathi-serve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/microsoft-sarathi-serve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=microsoft-sarathi-serve","description":"A low-latency & high-throughput serving engine for LLMs","homepage_url":null,"license":"Apache-2.0","open_issues":16,"watchers":7,"ai_summary":"Sarathi Serve is a Python-based project designed to provide efficient inference services with emphasis on low latency and high throughput for large language models.","readme_excerpt":"### Install Sarathi-Serve\n\n```sh\npip install -e .\n```","github_created_at":"2023-11-02T13:38:16+00:00","created_at":"2026-07-11T11:45:28.352442+00:00","updated_at":"2026-08-25T06:02:12.794444+00:00","categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"llama","name":"llama"},{"slug":"llm-inference","name":"llm-inference"},{"slug":"pytorch","name":"pytorch"},{"slug":"transformer","name":"transformer"}],"trust":{"provenance":{"is_fork":false,"github_id":713419018,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-08-25T06:02:11.930Z","maintenance":{"label":"Slowing","score":36,"methodology":"github_public_v1","releases_90d":0,"days_since_push":229,"last_release_at":null,"stars_delta_30d":8,"open_issues_delta_30d":0},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-11T11:45:29.669Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-08-25T06:02:12.430Z"},"languages":{"value":["python"],"source":"github.language+pyproject.toml","observed_at":"2026-08-25T06:02:12.430Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-08-25T06:02:12.430Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":null,"constraints":null,"when_to_use":["Optimize Python-based projects needing quick responses from large language models.","Demand high throughput alongside minimal inference delay."],"when_not_to_use":["Necessitate a non-Python environment for deployment and operation.","Prefer a tool that incorporates more than just low-latency, high-throughput focus such as multi-language support or specialized optimizations."],"source":"enrich:decision_facts","observed_at":"2026-07-17T05:16:53.426Z"},"constraint_facets":null,"decision_summary":[{"label":"Adopt for","value":"Sarathi Serve targets efficient low-latency and high-throughput inference for LLMs using Python."}]}}