{"data":{"node":{"slug":"deepspeedai-deepspeed-mii","name":"DeepSpeed-MII","tagline":"MII makes low-latency and high-throughput inference possible, powered by DeepSpeed.","github_url":"https://github.com/deepspeedai/DeepSpeed-MII","owner":"deepspeedai","repo":"DeepSpeed-MII","owner_avatar_url":"https://avatars.githubusercontent.com/u/74068820?v=4","primary_language":"Python","stars":2108,"forks":191,"topics":["deep-learning","inference","pytorch"],"archived":false,"github_pushed_at":"2025-06-30T16:21:45+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/deepspeedai-deepspeed-mii","markdown_url":"https://www.graphcanon.com/tools/deepspeedai-deepspeed-mii.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/deepspeedai-deepspeed-mii","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=deepspeedai-deepspeed-mii"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"deep-learning","name":"deep-learning"},{"slug":"inference","name":"inference"},{"slug":"pytorch","name":"pytorch"}],"edges":[],"neighbours":[{"slug":"deepspeedai-deepspeed","name":"DeepSpeed","tagline":"Deep learning optimization library for efficient distributed training and inference","github_url":"https://github.com/deepspeedai/DeepSpeed","owner":"deepspeedai","repo":"DeepSpeed","owner_avatar_url":"https://avatars.githubusercontent.com/u/74068820?v=4","primary_language":"Python","stars":42870,"forks":4920,"topics":["billion-parameters","compression","data-parallelism","deep-learning","gpu","inference","machine-learning","mixture-of-experts","model-parallelism","pipeline-parallelism","pytorch","trillion-parameters","zero"],"archived":false,"github_pushed_at":"2026-08-06T16:21:12+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/deepspeedai-deepspeed","markdown_url":"https://www.graphcanon.com/tools/deepspeedai-deepspeed.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/deepspeedai-deepspeed","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=deepspeedai-deepspeed","shared_categories":["inference-serving"]},{"slug":"ericlbuehler-mistral-rs","name":"mistral.rs","tagline":"Fast flexible LLM inference","github_url":"https://github.com/EricLBuehler/mistral.rs","owner":"EricLBuehler","repo":"mistral.rs","owner_avatar_url":"https://avatars.githubusercontent.com/u/65165915?v=4","primary_language":"Rust","stars":7575,"forks":671,"topics":["llm","rust","uqff"],"archived":false,"github_pushed_at":"2026-07-29T20:21:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs","markdown_url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ericlbuehler-mistral-rs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ericlbuehler-mistral-rs","shared_categories":["inference-serving"]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["inference-serving"]},{"slug":"jmaczan-tiny-vllm","name":"tiny-vllm","tagline":"Build your own high performance LLM inference engine in C++ and CUDA - a smaller version of vLLM","github_url":"https://github.com/jmaczan/tiny-vllm","owner":"jmaczan","repo":"tiny-vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/18054202?v=4","primary_language":"C++","stars":1075,"forks":84,"topics":["ai","attention","batching","course","cpp","cuda","hpc","inference","llm","llm-inference","pagedattention","tiny-vllm","vllm"],"archived":false,"github_pushed_at":"2026-08-23T14:20:13+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm","markdown_url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jmaczan-tiny-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jmaczan-tiny-vllm","shared_categories":["inference-serving"]},{"slug":"tensorchord-openmodelz","name":"openmodelz","tagline":"Automate and scale inference of large language models on Kubernetes.","github_url":"https://github.com/tensorchord/openmodelz","owner":"tensorchord","repo":"openmodelz","owner_avatar_url":"https://avatars.githubusercontent.com/u/100543303?v=4","primary_language":"Go","stars":282,"forks":26,"topics":["cluster-manager","hacktoberfest","inference","llm","llmops","mlops"],"archived":false,"github_pushed_at":"2023-11-03T06:33:25+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/tensorchord-openmodelz","markdown_url":"https://www.graphcanon.com/tools/tensorchord-openmodelz.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tensorchord-openmodelz","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tensorchord-openmodelz","shared_categories":["inference-serving"]},{"slug":"notai-tech-fastdeploy","name":"fastDeploy","tagline":"Deploy DL/ML inference pipelines with minimal extra code.","github_url":"https://github.com/notAI-tech/fastDeploy","owner":"notAI-tech","repo":"fastDeploy","owner_avatar_url":"https://avatars.githubusercontent.com/u/63401202?v=4","primary_language":"Python","stars":105,"forks":17,"topics":["deep-learning","docker","falcon","gevent","gunicorn","http-server","inference-server","model-deployment","model-serving","python","pytorch","serving","streaming-audio","tensorflow-serving","tf-serving","torchserve","triton","triton-inference-server","triton-server","websocket"],"archived":false,"github_pushed_at":"2026-02-10T16:18:52+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/notai-tech-fastdeploy","markdown_url":"https://www.graphcanon.com/tools/notai-tech-fastdeploy.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/notai-tech-fastdeploy","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=notai-tech-fastdeploy","shared_categories":["inference-serving"]}]}}