{"data":{"node":{"slug":"ai-hypercomputer-jetstream","name":"JetStream","tagline":"Throughput and memory optimized engine for LLM inference on XLA devices","github_url":"https://github.com/AI-Hypercomputer/JetStream","owner":"AI-Hypercomputer","repo":"JetStream","owner_avatar_url":"https://avatars.githubusercontent.com/u/181000646?v=4","primary_language":"Python","stars":455,"forks":67,"topics":["gemma","gpt","gpu","inference","jax","large-language-models","llama","llama2","llm","llm-inference","llmops","mlops","model-serving","pytorch","tpu","transformer"],"archived":false,"github_pushed_at":"2026-01-05T17:51:16+00:00","maintenance_label":"Slowing","stars_delta_30d":4,"url":"https://www.graphcanon.com/tools/ai-hypercomputer-jetstream","markdown_url":"https://www.graphcanon.com/tools/ai-hypercomputer-jetstream.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ai-hypercomputer-jetstream","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ai-hypercomputer-jetstream"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"gemma","name":"gemma"},{"slug":"gpt","name":"gpt"},{"slug":"gpu","name":"gpu"},{"slug":"inference","name":"inference"},{"slug":"jax","name":"jax"},{"slug":"large-language-models","name":"large language models"},{"slug":"llama","name":"llama"},{"slug":"llama2","name":"llama2"}],"edges":[],"neighbours":[{"slug":"vllm-project-vllm","name":"vllm","tagline":"A high-throughput and memory-efficient inference and serving engine for LLMs","github_url":"https://github.com/vllm-project/vllm","owner":"vllm-project","repo":"vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/136984999?v=4","primary_language":"Python","stars":87847,"forks":20135,"topics":["amd","blackwell","cuda","deepseek","deepseek-v3","gpt","gpt-oss","inference","kimi","llama","llm","llm-serving","model-serving","moe","openai","pytorch","qwen","qwen3","tpu","transformer"],"archived":false,"github_pushed_at":"2026-08-01T11:55:36+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/vllm-project-vllm","markdown_url":"https://www.graphcanon.com/tools/vllm-project-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vllm-project-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vllm-project-vllm","shared_categories":["inference-serving"]},{"slug":"lyogavin-airllm","name":"airllm","tagline":"AirLLM 70B inference with single 4GB GPU","github_url":"https://github.com/lyogavin/airllm","owner":"lyogavin","repo":"airllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/1113905?v=4","primary_language":"Jupyter Notebook","stars":24183,"forks":2722,"topics":["chinese-llm","chinese-nlp","finetune","generative-ai","instruct-gpt","instruction-set","llama","llm","lora","open-models","open-source","open-source-models","qlora"],"archived":false,"github_pushed_at":"2026-07-23T08:29:43+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lyogavin-airllm","markdown_url":"https://www.graphcanon.com/tools/lyogavin-airllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lyogavin-airllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lyogavin-airllm","shared_categories":["inference-serving"]},{"slug":"nvidia-tensorrt-llm","name":"TensorRT-LLM","tagline":"Python API for defining and optimizing Large Language Models (LLMs) on NVIDIA GPUs","github_url":"https://github.com/NVIDIA/TensorRT-LLM","owner":"NVIDIA","repo":"TensorRT-LLM","owner_avatar_url":"https://avatars.githubusercontent.com/u/1728152?v=4","primary_language":"Python","stars":14317,"forks":2641,"topics":["blackwell","cuda","llm-serving","moe","pytorch"],"archived":false,"github_pushed_at":"2026-08-07T05:40:26+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/nvidia-tensorrt-llm","markdown_url":"https://www.graphcanon.com/tools/nvidia-tensorrt-llm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-tensorrt-llm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-tensorrt-llm","shared_categories":["inference-serving"]},{"slug":"ericlbuehler-mistral-rs","name":"mistral.rs","tagline":"Fast flexible LLM inference","github_url":"https://github.com/EricLBuehler/mistral.rs","owner":"EricLBuehler","repo":"mistral.rs","owner_avatar_url":"https://avatars.githubusercontent.com/u/65165915?v=4","primary_language":"Rust","stars":7575,"forks":671,"topics":["llm","rust","uqff"],"archived":false,"github_pushed_at":"2026-07-29T20:21:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs","markdown_url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ericlbuehler-mistral-rs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ericlbuehler-mistral-rs","shared_categories":["inference-serving"]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["inference-serving"]},{"slug":"nvidia-transformerengine","name":"TransformerEngine","tagline":"A library for accelerating Transformer models on NVIDIA GPUs using low precision formats like FP8 and FP4.","github_url":"https://github.com/NVIDIA/TransformerEngine","owner":"NVIDIA","repo":"TransformerEngine","owner_avatar_url":"https://avatars.githubusercontent.com/u/1728152?v=4","primary_language":"Python","stars":3479,"forks":795,"topics":["cuda","deep-learning","fp4","fp8","gpu","jax","machine-learning","python","pytorch"],"archived":false,"github_pushed_at":"2026-08-07T05:44:50+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/nvidia-transformerengine","markdown_url":"https://www.graphcanon.com/tools/nvidia-transformerengine.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-transformerengine","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-transformerengine","shared_categories":["inference-serving"]},{"slug":"xllm-ai-xllm","name":"xllm","tagline":"A high-performance inference engine for LLM, VLM, DiT and REC models","github_url":"https://github.com/xLLM-AI/xllm","owner":"xLLM-AI","repo":"xllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/205719415?v=4","primary_language":"C++","stars":1534,"forks":282,"topics":["deepseek","glm","inference","inference-engine","large-language-models","llm-inference","qwen"],"archived":false,"github_pushed_at":"2026-08-24T09:51:10+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/xllm-ai-xllm","markdown_url":"https://www.graphcanon.com/tools/xllm-ai-xllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xllm-ai-xllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xllm-ai-xllm","shared_categories":["inference-serving"]},{"slug":"zhihu-zhilight","name":"ZhiLight","tagline":"A highly optimized LLM inference acceleration engine for Llama and its variants.","github_url":"https://github.com/zhihu/ZhiLight","owner":"zhihu","repo":"ZhiLight","owner_avatar_url":"https://avatars.githubusercontent.com/u/409513?v=4","primary_language":"C++","stars":908,"forks":104,"topics":["cuda","deepseek-r1","gpt","inference-engine","llama","llm","llm-inference","llm-serving","model-serving","pytorch"],"archived":false,"github_pushed_at":"2026-03-18T02:47:55+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/zhihu-zhilight","markdown_url":"https://www.graphcanon.com/tools/zhihu-zhilight.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zhihu-zhilight","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zhihu-zhilight","shared_categories":["inference-serving"]}]}}