{"data":{"node":{"slug":"avarok-cybersecurity-atlas","name":"atlas","tagline":"Pure Rust Inference Engine","github_url":"https://github.com/Avarok-Cybersecurity/atlas","owner":"Avarok-Cybersecurity","repo":"atlas","owner_avatar_url":"https://avatars.githubusercontent.com/u/96390941?v=4","primary_language":"Rust","stars":667,"forks":102,"topics":["cuda","dgx","dgx-spark","gb10","llm-inference","mamba","nvfp4","openai-api","rust","speculative-decoding","ssm","transformers"],"archived":false,"github_pushed_at":"2026-08-25T05:50:51+00:00","maintenance_label":"Very active","stars_delta_30d":57,"url":"https://www.graphcanon.com/tools/avarok-cybersecurity-atlas","markdown_url":"https://www.graphcanon.com/tools/avarok-cybersecurity-atlas.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/avarok-cybersecurity-atlas","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=avarok-cybersecurity-atlas"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"cuda","name":"cuda"},{"slug":"dgx","name":"dgx"},{"slug":"dgx-spark","name":"dgx-spark"},{"slug":"gb10","name":"gb10"},{"slug":"llm-inference","name":"llm-inference"},{"slug":"mamba","name":"mamba"},{"slug":"nvfp4","name":"nvfp4"},{"slug":"openai-api","name":"openai-api"}],"edges":[],"neighbours":[{"slug":"huggingface-candle","name":"candle","tagline":"Minimalist ML framework for Rust","github_url":"https://github.com/huggingface/candle","owner":"huggingface","repo":"candle","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Rust","stars":20825,"forks":1691,"topics":[],"archived":false,"github_pushed_at":"2026-07-30T06:00:09+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-candle","markdown_url":"https://www.graphcanon.com/tools/huggingface-candle.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-candle","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-candle","shared_categories":["inference-serving"]},{"slug":"jundot-omlx","name":"omlx","tagline":"LLM inference server with continuous batching and SSD caching for Apple Silicon","github_url":"https://github.com/jundot/omlx","owner":"jundot","repo":"omlx","owner_avatar_url":"https://avatars.githubusercontent.com/u/64250138?v=4","primary_language":"Python","stars":18679,"forks":1617,"topics":["apple-silicon","inference-server","llm","macos","mlx","openai-api"],"archived":false,"github_pushed_at":"2026-08-14T09:12:48+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/jundot-omlx","markdown_url":"https://www.graphcanon.com/tools/jundot-omlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jundot-omlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jundot-omlx","shared_categories":["inference-serving"]},{"slug":"ericlbuehler-mistral-rs","name":"mistral.rs","tagline":"Fast flexible LLM inference","github_url":"https://github.com/EricLBuehler/mistral.rs","owner":"EricLBuehler","repo":"mistral.rs","owner_avatar_url":"https://avatars.githubusercontent.com/u/65165915?v=4","primary_language":"Rust","stars":7575,"forks":671,"topics":["llm","rust","uqff"],"archived":false,"github_pushed_at":"2026-07-29T20:21:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs","markdown_url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ericlbuehler-mistral-rs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ericlbuehler-mistral-rs","shared_categories":["inference-serving"]},{"slug":"tensorflow-serving","name":"serving","tagline":"A flexible, high-performance serving system for machine learning models","github_url":"https://github.com/tensorflow/serving","owner":"tensorflow","repo":"serving","owner_avatar_url":"https://avatars.githubusercontent.com/u/15658638?v=4","primary_language":"C++","stars":6359,"forks":2204,"topics":["cpp","deep-learning","deep-neural-networks","machine-learning","ml","neural-network","python","serving","tensorflow"],"archived":false,"github_pushed_at":"2026-07-30T07:02:43+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/tensorflow-serving","markdown_url":"https://www.graphcanon.com/tools/tensorflow-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tensorflow-serving","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tensorflow-serving","shared_categories":["inference-serving"]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["inference-serving"]},{"slug":"xlite-dev-awesome-llm-inference","name":"Awesome-LLM-Inference","tagline":"A curated list of LLM/VLM inference papers with codes","github_url":"https://github.com/xlite-dev/Awesome-LLM-Inference","owner":"xlite-dev","repo":"Awesome-LLM-Inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/204302598?v=4","primary_language":"Python","stars":5477,"forks":429,"topics":["awesome-llm","deepseek","deepseek-r1","deepseek-v3","flash-attention","flash-attention-3","flash-mla","llm-inference","minimax-01","mla","paged-attention","qwen3","tensorrt-llm","vllm"],"archived":false,"github_pushed_at":"2026-08-14T12:23:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference","markdown_url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xlite-dev-awesome-llm-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xlite-dev-awesome-llm-inference","shared_categories":["inference-serving"]},{"slug":"b4rtaz-distributed-llama","name":"distributed-llama","tagline":"Distributed LLM inference using home devices cluster","github_url":"https://github.com/b4rtaz/distributed-llama","owner":"b4rtaz","repo":"distributed-llama","owner_avatar_url":"https://avatars.githubusercontent.com/u/12797776?v=4","primary_language":"C++","stars":3044,"forks":246,"topics":["distributed-computing","distributed-llm","llama2","llama3","llm","llm-inference","llms","neural-network","open-llm"],"archived":false,"github_pushed_at":"2026-07-05T16:47:20+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama","markdown_url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/b4rtaz-distributed-llama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=b4rtaz-distributed-llama","shared_categories":["inference-serving"]},{"slug":"michaelfeil-infinity","name":"infinity","tagline":"High-throughput, low-latency serving engine for text-embeddings and various models","github_url":"https://github.com/michaelfeil/infinity","owner":"michaelfeil","repo":"infinity","owner_avatar_url":"https://avatars.githubusercontent.com/u/63565275?v=4","primary_language":"Python","stars":2907,"forks":196,"topics":["bert-embeddings","llm","text-embeddings"],"archived":false,"github_pushed_at":"2026-03-24T03:59:47+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/michaelfeil-infinity","markdown_url":"https://www.graphcanon.com/tools/michaelfeil-infinity.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/michaelfeil-infinity","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=michaelfeil-infinity","shared_categories":["inference-serving"]},{"slug":"superlinked-sie","name":"sie","tagline":"Open-source inference server and production cluster for all the models your agent needs.","github_url":"https://github.com/superlinked/sie","owner":"superlinked","repo":"sie","owner_avatar_url":"https://avatars.githubusercontent.com/u/94243920?v=4","primary_language":"Python","stars":2804,"forks":272,"topics":["bge","colbert","data-pipeline","deep-learning","embeddings","inference","inference-server","information-retrieval","llm","ml","mlops","natural-language-processing","nlp","python","reranking","retrieval","retrieval-augmented-generation","semantic-search","splade","vector-search"],"archived":false,"github_pushed_at":"2026-08-21T20:28:04+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/superlinked-sie","markdown_url":"https://www.graphcanon.com/tools/superlinked-sie.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/superlinked-sie","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=superlinked-sie","shared_categories":["inference-serving"]},{"slug":"pykeio-ort","name":"ort","tagline":"Fast ML inference and training for ONNX models in Rust","github_url":"https://github.com/pykeio/ort","owner":"pykeio","repo":"ort","owner_avatar_url":"https://avatars.githubusercontent.com/u/62268720?v=4","primary_language":"Rust","stars":2472,"forks":263,"topics":["ai","ai-training","fine-tuning","inference","machine-learning","onnx","onnxruntime","rust"],"archived":false,"github_pushed_at":"2026-08-23T20:46:26+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/pykeio-ort","markdown_url":"https://www.graphcanon.com/tools/pykeio-ort.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pykeio-ort","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pykeio-ort","shared_categories":["inference-serving"]},{"slug":"abraxas-365-langchain-rust","name":"langchain-rust","tagline":"LangChain for Rust","github_url":"https://github.com/Abraxas-365/langchain-rust","owner":"Abraxas-365","repo":"langchain-rust","owner_avatar_url":"https://avatars.githubusercontent.com/u/63959220?v=4","primary_language":"Rust","stars":1339,"forks":176,"topics":["langchain","llm","llms","openai","rust"],"archived":false,"github_pushed_at":"2026-08-06T06:05:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/abraxas-365-langchain-rust","markdown_url":"https://www.graphcanon.com/tools/abraxas-365-langchain-rust.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/abraxas-365-langchain-rust","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=abraxas-365-langchain-rust","shared_categories":[]},{"slug":"anush008-fastembed-rs","name":"fastembed-rs","tagline":"Rust library for generating vector embeddings and reranking locally.","github_url":"https://github.com/Anush008/fastembed-rs","owner":"Anush008","repo":"fastembed-rs","owner_avatar_url":"https://avatars.githubusercontent.com/u/46051506?v=4","primary_language":"Rust","stars":992,"forks":136,"topics":["embeddings","fastembed","rag","reranker","reranking","retrieval","retrieval-augmented-generation","vector-search"],"archived":false,"github_pushed_at":"2026-08-16T17:50:30+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/anush008-fastembed-rs","markdown_url":"https://www.graphcanon.com/tools/anush008-fastembed-rs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/anush008-fastembed-rs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=anush008-fastembed-rs","shared_categories":[]}]}}