{"data":{"node":{"slug":"microsoft-minference","name":"MInference","tagline":"Accelerates Long-context LLMs' inference through approximate sparse calculation for attention.","github_url":"https://github.com/microsoft/MInference","owner":"microsoft","repo":"MInference","owner_avatar_url":"https://avatars.githubusercontent.com/u/6154722?v=4","primary_language":"Python","stars":1225,"forks":80,"topics":[],"archived":false,"github_pushed_at":"2026-04-08T08:04:38+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/microsoft-minference","markdown_url":"https://www.graphcanon.com/tools/microsoft-minference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/microsoft-minference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=microsoft-minference"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"attention-mechanism","name":"attention-mechanism"},{"slug":"flashattention-2","name":"flashattention-2"},{"slug":"inference-acceleration","name":"inference acceleration"},{"slug":"long-context-llms","name":"long-context llms"},{"slug":"sparse-calculation","name":"sparse calculation"},{"slug":"torch","name":"torch"}],"edges":[],"neighbours":[{"slug":"lyogavin-airllm","name":"airllm","tagline":"AirLLM 70B inference with single 4GB GPU","github_url":"https://github.com/lyogavin/airllm","owner":"lyogavin","repo":"airllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/1113905?v=4","primary_language":"Jupyter Notebook","stars":24183,"forks":2722,"topics":["chinese-llm","chinese-nlp","finetune","generative-ai","instruct-gpt","instruction-set","llama","llm","lora","open-models","open-source","open-source-models","qlora"],"archived":false,"github_pushed_at":"2026-07-23T08:29:43+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lyogavin-airllm","markdown_url":"https://www.graphcanon.com/tools/lyogavin-airllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lyogavin-airllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lyogavin-airllm","shared_categories":["inference-serving"]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["inference-serving"]},{"slug":"xlite-dev-awesome-llm-inference","name":"Awesome-LLM-Inference","tagline":"A curated list of LLM/VLM inference papers with codes","github_url":"https://github.com/xlite-dev/Awesome-LLM-Inference","owner":"xlite-dev","repo":"Awesome-LLM-Inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/204302598?v=4","primary_language":"Python","stars":5477,"forks":429,"topics":["awesome-llm","deepseek","deepseek-r1","deepseek-v3","flash-attention","flash-attention-3","flash-mla","llm-inference","minimax-01","mla","paged-attention","qwen3","tensorrt-llm","vllm"],"archived":false,"github_pushed_at":"2026-08-14T12:23:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference","markdown_url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xlite-dev-awesome-llm-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xlite-dev-awesome-llm-inference","shared_categories":["inference-serving"]},{"slug":"zhihu-zhilight","name":"ZhiLight","tagline":"A highly optimized LLM inference acceleration engine for Llama and its variants.","github_url":"https://github.com/zhihu/ZhiLight","owner":"zhihu","repo":"ZhiLight","owner_avatar_url":"https://avatars.githubusercontent.com/u/409513?v=4","primary_language":"C++","stars":908,"forks":104,"topics":["cuda","deepseek-r1","gpt","inference-engine","llama","llm","llm-inference","llm-serving","model-serving","pytorch"],"archived":false,"github_pushed_at":"2026-03-18T02:47:55+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/zhihu-zhilight","markdown_url":"https://www.graphcanon.com/tools/zhihu-zhilight.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zhihu-zhilight","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zhihu-zhilight","shared_categories":["inference-serving"]},{"slug":"feifeibear-long-context-attention","name":"long-context-attention","tagline":"Unified Sequence Parallel Attention for Long Context Transformers","github_url":"https://github.com/feifeibear/long-context-attention","owner":"feifeibear","repo":"long-context-attention","owner_avatar_url":"https://avatars.githubusercontent.com/u/5706969?v=4","primary_language":"Python","stars":687,"forks":83,"topics":["attention-is-all-you-need","deepspeed-ulysses","llm-inference","llm-training","pytorch","ring-attention"],"archived":false,"github_pushed_at":"2026-05-21T06:53:42+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/feifeibear-long-context-attention","markdown_url":"https://www.graphcanon.com/tools/feifeibear-long-context-attention.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/feifeibear-long-context-attention","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=feifeibear-long-context-attention","shared_categories":["inference-serving"]},{"slug":"quantumaikr-quant-cpp","name":"quant.cpp","tagline":"LLM inference with extended context using C","github_url":"https://github.com/quantumaikr/quant.cpp","owner":"quantumaikr","repo":"quant.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/49489033?v=4","primary_language":"C","stars":399,"forks":44,"topics":["delta-compression","embeddable","gguf","kv-cache","llm","llm-inference","pure-c","quantization","transformer","turboquant"],"archived":false,"github_pushed_at":"2026-04-26T11:15:51+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/quantumaikr-quant-cpp","markdown_url":"https://www.graphcanon.com/tools/quantumaikr-quant-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/quantumaikr-quant-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=quantumaikr-quant-cpp","shared_categories":["inference-serving"]},{"slug":"nvidia-star-attention","name":"Star-Attention","tagline":"Efficient LLM Inference over Long Sequences","github_url":"https://github.com/NVIDIA/Star-Attention","owner":"NVIDIA","repo":"Star-Attention","owner_avatar_url":"https://avatars.githubusercontent.com/u/1728152?v=4","primary_language":"Python","stars":392,"forks":25,"topics":["attention-mechanism","large-language-models","llm-inference"],"archived":false,"github_pushed_at":"2025-06-25T19:36:21+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/nvidia-star-attention","markdown_url":"https://www.graphcanon.com/tools/nvidia-star-attention.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-star-attention","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-star-attention","shared_categories":["inference-serving"]}]}}