{"data":{"node":{"slug":"vllm-project-semantic-router","name":"semantic-router","tagline":"System level intelligent runtime for Mixture-of-Models across edge, data center and cloud","github_url":"https://github.com/vllm-project/semantic-router","owner":"vllm-project","repo":"semantic-router","owner_avatar_url":"https://avatars.githubusercontent.com/u/136984999?v=4","primary_language":"Go","stars":5241,"forks":825,"topics":["ai-gateway","bert-classification","fine-tuning","golang","huggingface-candle","huggingface-transformers","kubernetes","llm","llmrouter","mixture-of-models","pii-detection","prompt-engineering","prompt-guard","rust","semantic-router","vllm"],"archived":false,"github_pushed_at":"2026-08-23T17:59:46+00:00","maintenance_label":"Very active","stars_delta_30d":202,"url":"https://www.graphcanon.com/tools/vllm-project-semantic-router","markdown_url":"https://www.graphcanon.com/tools/vllm-project-semantic-router.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vllm-project-semantic-router","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vllm-project-semantic-router"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"},{"slug":"llm-frameworks","name":"LLM Frameworks","url":"https://www.graphcanon.com/categories/llm-frameworks","markdown_url":"https://www.graphcanon.com/categories/llm-frameworks.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/llm-frameworks"}],"tags":[{"slug":"ai-gateway","name":"ai-gateway"},{"slug":"bert-classification","name":"bert-classification"},{"slug":"fine-tuning","name":"fine-tuning"},{"slug":"golang","name":"golang"},{"slug":"huggingface-candle","name":"huggingface-candle"},{"slug":"huggingface-transformers","name":"huggingface-transformers"},{"slug":"kubernetes","name":"kubernetes"},{"slug":"llmrouter","name":"llmrouter"}],"edges":[],"neighbours":[{"slug":"portkey-ai-gateway","name":"gateway","tagline":"A high-performance AI Gateway connecting to over 1,600 LLMs with guardrails.","github_url":"https://github.com/Portkey-AI/gateway","owner":"Portkey-AI","repo":"gateway","owner_avatar_url":"https://avatars.githubusercontent.com/u/131141116?v=4","primary_language":"TypeScript","stars":12668,"forks":1236,"topics":["ai-gateway","gateway","generative-ai","hacktoberfest","langchain","llm","llm-gateway","llmops","llms","mcp","mcp-client","mcp-gateway","mcp-servers","model-router","openai"],"archived":false,"github_pushed_at":"2026-05-25T13:54:51+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/portkey-ai-gateway","markdown_url":"https://www.graphcanon.com/tools/portkey-ai-gateway.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/portkey-ai-gateway","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=portkey-ai-gateway","shared_categories":["llm-frameworks"]},{"slug":"ai-dynamo-dynamo","name":"dynamo","tagline":"A Datacenter Scale Distributed Inference Serving Framework","github_url":"https://github.com/ai-dynamo/dynamo","owner":"ai-dynamo","repo":"dynamo","owner_avatar_url":"https://avatars.githubusercontent.com/u/201626793?v=4","primary_language":"Rust","stars":7845,"forks":1486,"topics":["diffusion","disaggregated-serving","kubernetes","llm-inference","omni","routing-engine","rust","sglang","tensorrt-llm","vllm"],"archived":false,"github_pushed_at":"2026-08-24T17:59:43+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/ai-dynamo-dynamo","markdown_url":"https://www.graphcanon.com/tools/ai-dynamo-dynamo.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ai-dynamo-dynamo","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ai-dynamo-dynamo","shared_categories":["inference-serving"]},{"slug":"ericlbuehler-mistral-rs","name":"mistral.rs","tagline":"Fast flexible LLM inference","github_url":"https://github.com/EricLBuehler/mistral.rs","owner":"EricLBuehler","repo":"mistral.rs","owner_avatar_url":"https://avatars.githubusercontent.com/u/65165915?v=4","primary_language":"Rust","stars":7575,"forks":671,"topics":["llm","rust","uqff"],"archived":false,"github_pushed_at":"2026-07-29T20:21:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs","markdown_url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ericlbuehler-mistral-rs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ericlbuehler-mistral-rs","shared_categories":["inference-serving"]},{"slug":"tensorflow-serving","name":"serving","tagline":"A flexible, high-performance serving system for machine learning models","github_url":"https://github.com/tensorflow/serving","owner":"tensorflow","repo":"serving","owner_avatar_url":"https://avatars.githubusercontent.com/u/15658638?v=4","primary_language":"C++","stars":6359,"forks":2204,"topics":["cpp","deep-learning","deep-neural-networks","machine-learning","ml","neural-network","python","serving","tensorflow"],"archived":false,"github_pushed_at":"2026-07-30T07:02:43+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/tensorflow-serving","markdown_url":"https://www.graphcanon.com/tools/tensorflow-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tensorflow-serving","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tensorflow-serving","shared_categories":["inference-serving"]},{"slug":"seldonio-seldon-core","name":"seldon-core","tagline":"An MLOps framework to package, deploy, monitor and manage thousands of production machine learning models","github_url":"https://github.com/SeldonIO/seldon-core","owner":"SeldonIO","repo":"seldon-core","owner_avatar_url":"https://avatars.githubusercontent.com/u/10297834?v=4","primary_language":"Go","stars":4765,"forks":867,"topics":["aiops","deployment","kubernetes","machine-learning","machine-learning-operations","mlops","production-machine-learning","serving"],"archived":false,"github_pushed_at":"2026-03-23T11:39:54+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/seldonio-seldon-core","markdown_url":"https://www.graphcanon.com/tools/seldonio-seldon-core.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/seldonio-seldon-core","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=seldonio-seldon-core","shared_categories":["inference-serving"]},{"slug":"huaizhengzhang-ai-infra-from-zero-to-hero","name":"AI-Infra-from-Zero-to-Hero","tagline":"Awesome System for Machine Learning and LLM Infra","github_url":"https://github.com/HuaizhengZhang/AI-Infra-from-Zero-to-Hero","owner":"HuaizhengZhang","repo":"AI-Infra-from-Zero-to-Hero","owner_avatar_url":"https://avatars.githubusercontent.com/u/5894780?v=4","primary_language":null,"stars":4285,"forks":409,"topics":["ai-infra","genai","large-language-models","llmsys","mlsys","model-serving","model-training"],"archived":false,"github_pushed_at":"2025-07-25T02:24:35+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/huaizhengzhang-ai-infra-from-zero-to-hero","markdown_url":"https://www.graphcanon.com/tools/huaizhengzhang-ai-infra-from-zero-to-hero.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huaizhengzhang-ai-infra-from-zero-to-hero","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huaizhengzhang-ai-infra-from-zero-to-hero","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"a16z-infra-ai-getting-started","name":"ai-getting-started","tagline":"A Javascript AI getting started stack for weekend projects","github_url":"https://github.com/a16z-infra/ai-getting-started","owner":"a16z-infra","repo":"ai-getting-started","owner_avatar_url":"https://avatars.githubusercontent.com/u/130202746?v=4","primary_language":"TypeScript","stars":4141,"forks":660,"topics":[],"archived":false,"github_pushed_at":"2024-08-21T12:35:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/a16z-infra-ai-getting-started","markdown_url":"https://www.graphcanon.com/tools/a16z-infra-ai-getting-started.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/a16z-infra-ai-getting-started","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=a16z-infra-ai-getting-started","shared_categories":[]},{"slug":"aurelio-labs-semantic-router","name":"semantic-router","tagline":"Superfast AI decision making and intelligent processing of multi-modal data","github_url":"https://github.com/aurelio-labs/semantic-router","owner":"aurelio-labs","repo":"semantic-router","owner_avatar_url":"https://avatars.githubusercontent.com/u/69076224?v=4","primary_language":"Python","stars":3764,"forks":358,"topics":["ai","artificial-intelligence","chatbot","computer-vision","generative-ai","machine-learning","nlp"],"archived":false,"github_pushed_at":"2026-07-26T23:06:25+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/aurelio-labs-semantic-router","markdown_url":"https://www.graphcanon.com/tools/aurelio-labs-semantic-router.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/aurelio-labs-semantic-router","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=aurelio-labs-semantic-router","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"b4rtaz-distributed-llama","name":"distributed-llama","tagline":"Distributed LLM inference using home devices cluster","github_url":"https://github.com/b4rtaz/distributed-llama","owner":"b4rtaz","repo":"distributed-llama","owner_avatar_url":"https://avatars.githubusercontent.com/u/12797776?v=4","primary_language":"C++","stars":3044,"forks":246,"topics":["distributed-computing","distributed-llm","llama2","llama3","llm","llm-inference","llms","neural-network","open-llm"],"archived":false,"github_pushed_at":"2026-07-05T16:47:20+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama","markdown_url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/b4rtaz-distributed-llama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=b4rtaz-distributed-llama","shared_categories":["inference-serving"]},{"slug":"michaelfeil-infinity","name":"infinity","tagline":"High-throughput, low-latency serving engine for text-embeddings and various models","github_url":"https://github.com/michaelfeil/infinity","owner":"michaelfeil","repo":"infinity","owner_avatar_url":"https://avatars.githubusercontent.com/u/63565275?v=4","primary_language":"Python","stars":2907,"forks":196,"topics":["bert-embeddings","llm","text-embeddings"],"archived":false,"github_pushed_at":"2026-03-24T03:59:47+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/michaelfeil-infinity","markdown_url":"https://www.graphcanon.com/tools/michaelfeil-infinity.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/michaelfeil-infinity","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=michaelfeil-infinity","shared_categories":["inference-serving"]},{"slug":"superlinked-sie","name":"sie","tagline":"Open-source inference server and production cluster for all the models your agent needs.","github_url":"https://github.com/superlinked/sie","owner":"superlinked","repo":"sie","owner_avatar_url":"https://avatars.githubusercontent.com/u/94243920?v=4","primary_language":"Python","stars":2804,"forks":272,"topics":["bge","colbert","data-pipeline","deep-learning","embeddings","inference","inference-server","information-retrieval","llm","ml","mlops","natural-language-processing","nlp","python","reranking","retrieval","retrieval-augmented-generation","semantic-search","splade","vector-search"],"archived":false,"github_pushed_at":"2026-08-21T20:28:04+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/superlinked-sie","markdown_url":"https://www.graphcanon.com/tools/superlinked-sie.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/superlinked-sie","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=superlinked-sie","shared_categories":["inference-serving"]},{"slug":"vercel-modelfusion","name":"modelfusion","tagline":"TypeScript library for building AI applications","github_url":"https://github.com/vercel/modelfusion","owner":"vercel","repo":"modelfusion","owner_avatar_url":"https://avatars.githubusercontent.com/u/14985020?v=4","primary_language":"TypeScript","stars":1319,"forks":95,"topics":["ai","artificial-intelligence","chatbot","claude","dall-e","embedding","gpt-3","huggingface","javascript","js","llamacpp","llm","mistral","multi-modal","ollama","openai","stable-diffusion","ts","typescript","whisper"],"archived":true,"github_pushed_at":"2024-07-19T15:17:19+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/vercel-modelfusion","markdown_url":"https://www.graphcanon.com/tools/vercel-modelfusion.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vercel-modelfusion","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vercel-modelfusion","shared_categories":["llm-frameworks"]}]}}