{"data":{"node":{"slug":"kaden-schutt-hipfire","name":"hipfire","tagline":"RDNA-native LLM inference engine in Rust","github_url":"https://github.com/Kaden-Schutt/hipfire","owner":"Kaden-Schutt","repo":"hipfire","owner_avatar_url":"https://avatars.githubusercontent.com/u/151092359?v=4","primary_language":"Rust","stars":554,"forks":61,"topics":["amd-gpu","gpu-computing","hip","llm-inference","machine-learning","quantization","rdna","rocm","rust"],"archived":false,"github_pushed_at":"2026-08-25T03:19:07+00:00","maintenance_label":"Very active","stars_delta_30d":63,"url":"https://www.graphcanon.com/tools/kaden-schutt-hipfire","markdown_url":"https://www.graphcanon.com/tools/kaden-schutt-hipfire.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kaden-schutt-hipfire","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kaden-schutt-hipfire"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"amd-gpu","name":"amd-gpu"},{"slug":"gpu-computing","name":"gpu-computing"},{"slug":"hip","name":"hip"},{"slug":"llm-inference","name":"llm-inference"},{"slug":"machine-learning","name":"machine-learning"},{"slug":"quantization","name":"quantization"},{"slug":"rdna","name":"rdna"},{"slug":"rocm","name":"rocm"}],"edges":[],"neighbours":[{"slug":"alexsjones-llmfit","name":"llmfit","tagline":"Hundreds of models & providers. One command to find what runs on your hardware.","github_url":"https://github.com/AlexsJones/llmfit","owner":"AlexsJones","repo":"llmfit","owner_avatar_url":"https://avatars.githubusercontent.com/u/1235925?v=4","primary_language":"Rust","stars":31867,"forks":1978,"topics":["gguf","llm","localai","mlx","skill","unsloth"],"archived":false,"github_pushed_at":"2026-08-14T07:36:41+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/alexsjones-llmfit","markdown_url":"https://www.graphcanon.com/tools/alexsjones-llmfit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alexsjones-llmfit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alexsjones-llmfit","shared_categories":[]},{"slug":"huggingface-candle","name":"candle","tagline":"Minimalist ML framework for Rust","github_url":"https://github.com/huggingface/candle","owner":"huggingface","repo":"candle","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Rust","stars":20825,"forks":1691,"topics":[],"archived":false,"github_pushed_at":"2026-07-30T06:00:09+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-candle","markdown_url":"https://www.graphcanon.com/tools/huggingface-candle.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-candle","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-candle","shared_categories":["inference-serving"]},{"slug":"ericlbuehler-mistral-rs","name":"mistral.rs","tagline":"Fast flexible LLM inference","github_url":"https://github.com/EricLBuehler/mistral.rs","owner":"EricLBuehler","repo":"mistral.rs","owner_avatar_url":"https://avatars.githubusercontent.com/u/65165915?v=4","primary_language":"Rust","stars":7575,"forks":671,"topics":["llm","rust","uqff"],"archived":false,"github_pushed_at":"2026-07-29T20:21:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs","markdown_url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ericlbuehler-mistral-rs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ericlbuehler-mistral-rs","shared_categories":["inference-serving"]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["inference-serving"]},{"slug":"michael-a-kuykendall-shimmy","name":"shimmy","tagline":"⚡ A Pure-Rust WebGPU Inference Engine, OpenAI-API Compatible and Native to GGUF","github_url":"https://github.com/Michael-A-Kuykendall/shimmy","owner":"Michael-A-Kuykendall","repo":"shimmy","owner_avatar_url":"https://avatars.githubusercontent.com/u/196678413?v=4","primary_language":"Rust","stars":5808,"forks":559,"topics":["api-server","command-line-tool","developer-tools","gguf","huggingface","huggingface-models","huggingface-transformers","inference-server","llama","llamacpp","llm-inference","local-ai","machine-learning","ollama-api","openai-compatible","rust","rust-crate","transformers","webgpu","webgpu-shaders"],"archived":false,"github_pushed_at":"2026-08-20T17:39:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/michael-a-kuykendall-shimmy","markdown_url":"https://www.graphcanon.com/tools/michael-a-kuykendall-shimmy.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/michael-a-kuykendall-shimmy","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=michael-a-kuykendall-shimmy","shared_categories":["inference-serving"]},{"slug":"b4rtaz-distributed-llama","name":"distributed-llama","tagline":"Distributed LLM inference using home devices cluster","github_url":"https://github.com/b4rtaz/distributed-llama","owner":"b4rtaz","repo":"distributed-llama","owner_avatar_url":"https://avatars.githubusercontent.com/u/12797776?v=4","primary_language":"C++","stars":3044,"forks":246,"topics":["distributed-computing","distributed-llm","llama2","llama3","llm","llm-inference","llms","neural-network","open-llm"],"archived":false,"github_pushed_at":"2026-07-05T16:47:20+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama","markdown_url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/b4rtaz-distributed-llama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=b4rtaz-distributed-llama","shared_categories":["inference-serving"]},{"slug":"michaelfeil-infinity","name":"infinity","tagline":"High-throughput, low-latency serving engine for text-embeddings and various models","github_url":"https://github.com/michaelfeil/infinity","owner":"michaelfeil","repo":"infinity","owner_avatar_url":"https://avatars.githubusercontent.com/u/63565275?v=4","primary_language":"Python","stars":2907,"forks":196,"topics":["bert-embeddings","llm","text-embeddings"],"archived":false,"github_pushed_at":"2026-03-24T03:59:47+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/michaelfeil-infinity","markdown_url":"https://www.graphcanon.com/tools/michaelfeil-infinity.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/michaelfeil-infinity","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=michaelfeil-infinity","shared_categories":["inference-serving"]},{"slug":"pykeio-ort","name":"ort","tagline":"Fast ML inference and training for ONNX models in Rust","github_url":"https://github.com/pykeio/ort","owner":"pykeio","repo":"ort","owner_avatar_url":"https://avatars.githubusercontent.com/u/62268720?v=4","primary_language":"Rust","stars":2472,"forks":263,"topics":["ai","ai-training","fine-tuning","inference","machine-learning","onnx","onnxruntime","rust"],"archived":false,"github_pushed_at":"2026-08-23T20:46:26+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/pykeio-ort","markdown_url":"https://www.graphcanon.com/tools/pykeio-ort.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pykeio-ort","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pykeio-ort","shared_categories":["inference-serving"]},{"slug":"abraxas-365-langchain-rust","name":"langchain-rust","tagline":"LangChain for Rust","github_url":"https://github.com/Abraxas-365/langchain-rust","owner":"Abraxas-365","repo":"langchain-rust","owner_avatar_url":"https://avatars.githubusercontent.com/u/63959220?v=4","primary_language":"Rust","stars":1339,"forks":176,"topics":["langchain","llm","llms","openai","rust"],"archived":false,"github_pushed_at":"2026-08-06T06:05:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/abraxas-365-langchain-rust","markdown_url":"https://www.graphcanon.com/tools/abraxas-365-langchain-rust.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/abraxas-365-langchain-rust","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=abraxas-365-langchain-rust","shared_categories":[]},{"slug":"jmaczan-tiny-vllm","name":"tiny-vllm","tagline":"Build your own high performance LLM inference engine in C++ and CUDA - a smaller version of vLLM","github_url":"https://github.com/jmaczan/tiny-vllm","owner":"jmaczan","repo":"tiny-vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/18054202?v=4","primary_language":"C++","stars":1075,"forks":84,"topics":["ai","attention","batching","course","cpp","cuda","hpc","inference","llm","llm-inference","pagedattention","tiny-vllm","vllm"],"archived":false,"github_pushed_at":"2026-08-23T14:20:13+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm","markdown_url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jmaczan-tiny-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jmaczan-tiny-vllm","shared_categories":["inference-serving"]},{"slug":"keyvank-femtogpt","name":"femtoGPT","tagline":"Pure Rust implementation of a minimal Generative Pretrained Transformer","github_url":"https://github.com/keyvank/femtoGPT","owner":"keyvank","repo":"femtoGPT","owner_avatar_url":"https://avatars.githubusercontent.com/u/4275654?v=4","primary_language":"Rust","stars":935,"forks":67,"topics":["from-scratch","gpt","gpu","hacktoberfest","llm","machine-learning","neural-network","opencl","rust"],"archived":false,"github_pushed_at":"2025-10-21T11:13:42+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/keyvank-femtogpt","markdown_url":"https://www.graphcanon.com/tools/keyvank-femtogpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/keyvank-femtogpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=keyvank-femtogpt","shared_categories":[]},{"slug":"harleyszhang-llm-note","name":"llm_note","tagline":"LLM notes covering model inference transformer structures and framework analysis","github_url":"https://github.com/harleyszhang/llm_note","owner":"harleyszhang","repo":"llm_note","owner_avatar_url":"https://avatars.githubusercontent.com/u/37138671?v=4","primary_language":"Python","stars":888,"forks":90,"topics":["cuda-programming","kv-cache","llm","llm-inference","transformer-models","triton-kernels","vllm"],"archived":false,"github_pushed_at":"2026-08-19T06:46:41+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/harleyszhang-llm-note","markdown_url":"https://www.graphcanon.com/tools/harleyszhang-llm-note.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/harleyszhang-llm-note","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=harleyszhang-llm-note","shared_categories":["inference-serving"]}]}}