{"data":{"node":{"slug":"vertexclique-orkhon","name":"orkhon","tagline":"ML Inference Framework and Server Runtime","github_url":"https://github.com/vertexclique/orkhon","owner":"vertexclique","repo":"orkhon","owner_avatar_url":"https://avatars.githubusercontent.com/u/578559?v=4","primary_language":"Rust","stars":153,"forks":4,"topics":["async","data-parallelism","inference-server","machine-learning","multiprocessing","python3","tensorflow"],"archived":false,"github_pushed_at":"2021-02-01T15:23:47+00:00","maintenance_label":"Dormant","stars_delta_30d":0,"url":"https://www.graphcanon.com/tools/vertexclique-orkhon","markdown_url":"https://www.graphcanon.com/tools/vertexclique-orkhon.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vertexclique-orkhon","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vertexclique-orkhon"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"async","name":"async"},{"slug":"data-parallelism","name":"data-parallelism"},{"slug":"multiprocessing","name":"multiprocessing"},{"slug":"python3","name":"python3"},{"slug":"tensorflow","name":"tensorflow"}],"edges":[],"neighbours":[{"slug":"jundot-omlx","name":"omlx","tagline":"LLM inference server with continuous batching and SSD caching for Apple Silicon","github_url":"https://github.com/jundot/omlx","owner":"jundot","repo":"omlx","owner_avatar_url":"https://avatars.githubusercontent.com/u/64250138?v=4","primary_language":"Python","stars":21934,"forks":1899,"topics":["apple-silicon","inference-server","llm","macos","mlx","openai-api"],"archived":false,"github_pushed_at":"2026-09-20T02:23:26+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jundot-omlx","markdown_url":"https://www.graphcanon.com/tools/jundot-omlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jundot-omlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jundot-omlx","shared_categories":["inference-serving"]},{"slug":"ai-dynamo-dynamo","name":"dynamo","tagline":"A Datacenter Scale Distributed Inference Serving Framework","github_url":"https://github.com/ai-dynamo/dynamo","owner":"ai-dynamo","repo":"dynamo","owner_avatar_url":"https://avatars.githubusercontent.com/u/201626793?v=4","primary_language":"Rust","stars":8121,"forks":1601,"topics":["diffusion","disaggregated-serving","kubernetes","llm-inference","omni","routing-engine","rust","sglang","tensorrt-llm","vllm"],"archived":false,"github_pushed_at":"2026-09-19T17:01:54+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/ai-dynamo-dynamo","markdown_url":"https://www.graphcanon.com/tools/ai-dynamo-dynamo.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ai-dynamo-dynamo","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ai-dynamo-dynamo","shared_categories":["inference-serving"]},{"slug":"ericlbuehler-mistral-rs","name":"mistral.rs","tagline":"Fast flexible LLM inference","github_url":"https://github.com/EricLBuehler/mistral.rs","owner":"EricLBuehler","repo":"mistral.rs","owner_avatar_url":"https://avatars.githubusercontent.com/u/65165915?v=4","primary_language":"Rust","stars":7660,"forks":690,"topics":["llm","rust","uqff"],"archived":false,"github_pushed_at":"2026-09-06T20:01:45+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs","markdown_url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ericlbuehler-mistral-rs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ericlbuehler-mistral-rs","shared_categories":["inference-serving"]},{"slug":"linkedin-liger-kernel","name":"Liger-Kernel","tagline":"Efficient Triton Kernels for LLM Training","github_url":"https://github.com/linkedin/Liger-Kernel","owner":"linkedin","repo":"Liger-Kernel","owner_avatar_url":"https://avatars.githubusercontent.com/u/357098?v=4","primary_language":"Python","stars":6600,"forks":593,"topics":["finetuning","gemma2","hacktoberfest","llama","llama3","llm-training","llms","mistral","phi3","triton","triton-kernels"],"archived":false,"github_pushed_at":"2026-09-03T15:56:18+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/linkedin-liger-kernel","markdown_url":"https://www.graphcanon.com/tools/linkedin-liger-kernel.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/linkedin-liger-kernel","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=linkedin-liger-kernel","shared_categories":[]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Cuda","stars":6452,"forks":1472,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-09-19T01:31:07+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["inference-serving"]},{"slug":"kserve-kserve","name":"kserve","tagline":"Standardized Distributed Generative and Predictive AI Inference Platform for Scalable, Multi-Framework Deployment on Kubernetes","github_url":"https://github.com/kserve/kserve","owner":"kserve","repo":"kserve","owner_avatar_url":"https://avatars.githubusercontent.com/u/83512434?v=4","primary_language":"Go","stars":5970,"forks":1689,"topics":["artificial-intelligence","cncf","genai","hacktoberfest","istio","k8s","knative","kserve","kubeflow","kubernetes","llm-inference","machine-learning","mlops","model-interpretability","model-serving","pytorch","service-mesh","tensorflow","vllm","xgboost"],"archived":false,"github_pushed_at":"2026-09-19T13:08:33+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/kserve-kserve","markdown_url":"https://www.graphcanon.com/tools/kserve-kserve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kserve-kserve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kserve-kserve","shared_categories":["inference-serving"]},{"slug":"vllm-project-semantic-router","name":"semantic-router","tagline":"System level intelligent runtime for Mixture-of-Models across edge, data center and cloud","github_url":"https://github.com/vllm-project/semantic-router","owner":"vllm-project","repo":"semantic-router","owner_avatar_url":"https://avatars.githubusercontent.com/u/136984999?v=4","primary_language":"Go","stars":5869,"forks":957,"topics":["ai-gateway","guardrails","inference","kubernetes","llm","llmrouter","mixture-of-models","pytorch","semantic-router","transformer","vllm"],"archived":false,"github_pushed_at":"2026-09-19T18:18:46+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/vllm-project-semantic-router","markdown_url":"https://www.graphcanon.com/tools/vllm-project-semantic-router.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vllm-project-semantic-router","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vllm-project-semantic-router","shared_categories":["inference-serving"]},{"slug":"pytorch-serve","name":"serve","tagline":"Serve, optimize and scale PyTorch models in production","github_url":"https://github.com/pytorch/serve","owner":"pytorch","repo":"serve","owner_avatar_url":"https://avatars.githubusercontent.com/u/21003710?v=4","primary_language":"Java","stars":4344,"forks":880,"topics":["cpu","deep-learning","docker","gpu","kubernetes","machine-learning","metrics","mlops","optimization","pytorch","serving"],"archived":true,"github_pushed_at":"2025-08-06T19:17:08+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/pytorch-serve","markdown_url":"https://www.graphcanon.com/tools/pytorch-serve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pytorch-serve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pytorch-serve","shared_categories":["inference-serving"]},{"slug":"superlinked-sie","name":"sie","tagline":"Open-source inference server and production cluster for all the models your agent needs.","github_url":"https://github.com/superlinked/sie","owner":"superlinked","repo":"sie","owner_avatar_url":"https://avatars.githubusercontent.com/u/94243920?v=4","primary_language":"Python","stars":3293,"forks":311,"topics":["bge","colbert","data-pipeline","deep-learning","embeddings","inference","inference-server","information-retrieval","llm","ml","mlops","natural-language-processing","nlp","python","reranking","retrieval","retrieval-augmented-generation","semantic-search","splade","vector-search"],"archived":false,"github_pushed_at":"2026-09-19T01:00:02+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/superlinked-sie","markdown_url":"https://www.graphcanon.com/tools/superlinked-sie.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/superlinked-sie","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=superlinked-sie","shared_categories":["inference-serving"]},{"slug":"b4rtaz-distributed-llama","name":"distributed-llama","tagline":"Distributed LLM inference using home devices cluster","github_url":"https://github.com/b4rtaz/distributed-llama","owner":"b4rtaz","repo":"distributed-llama","owner_avatar_url":"https://avatars.githubusercontent.com/u/12797776?v=4","primary_language":"C++","stars":3060,"forks":250,"topics":["distributed-computing","distributed-llm","llama2","llama3","llm","llm-inference","llms","neural-network","open-llm"],"archived":false,"github_pushed_at":"2026-07-05T16:47:20+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama","markdown_url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/b4rtaz-distributed-llama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=b4rtaz-distributed-llama","shared_categories":["inference-serving"]},{"slug":"microsoft-pai","name":"pai","tagline":"Resource scheduling and cluster management for AI","github_url":"https://github.com/microsoft/pai","owner":"microsoft","repo":"pai","owner_avatar_url":"https://avatars.githubusercontent.com/u/6154722?v=4","primary_language":"JavaScript","stars":2688,"forks":552,"topics":["ai","artificial-intelligence","chainer","cloud","cluster-management","cluster-manager","gpu","gpu-cluster","gpu-computing","gpu-scheduler","jupyter","kubernetes","machine-learning","model-training","on-premise","pytorch","resource-management","scheduling","tensorflow"],"archived":false,"github_pushed_at":"2026-08-15T00:09:09+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/microsoft-pai","markdown_url":"https://www.graphcanon.com/tools/microsoft-pai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/microsoft-pai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=microsoft-pai","shared_categories":["inference-serving"]},{"slug":"pykeio-ort","name":"ort","tagline":"Fast ML inference and training for ONNX models in Rust","github_url":"https://github.com/pykeio/ort","owner":"pykeio","repo":"ort","owner_avatar_url":"https://avatars.githubusercontent.com/u/62268720?v=4","primary_language":"Rust","stars":2515,"forks":271,"topics":["ai","ai-training","fine-tuning","inference","machine-learning","onnx","onnxruntime","rust"],"archived":false,"github_pushed_at":"2026-09-16T09:56:02+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/pykeio-ort","markdown_url":"https://www.graphcanon.com/tools/pykeio-ort.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pykeio-ort","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pykeio-ort","shared_categories":["inference-serving"]}]}}