{"data":{"node":{"slug":"rubixml-server","name":"Server","tagline":"Standalone inference server for Rubix ML estimators.","github_url":"https://github.com/RubixML/Server","owner":"RubixML","repo":"Server","owner_avatar_url":"https://avatars.githubusercontent.com/u/43308973?v=4","primary_language":"PHP","stars":63,"forks":13,"topics":["api","http-server","inference","inference-engine","inference-server","infrastructure","json-api","machine-learning","microservice","ml-infrastructure","model-deployment","model-server","php","php-machine-learning","php-ml","rest-api","rubix-ml","rubix-server"],"archived":false,"github_pushed_at":"2026-03-03T01:42:30+00:00","maintenance_label":"Slowing","stars_delta_30d":0,"url":"https://www.graphcanon.com/tools/rubixml-server","markdown_url":"https://www.graphcanon.com/tools/rubixml-server.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rubixml-server","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rubixml-server"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"api","name":"api"},{"slug":"http-server","name":"http-server"},{"slug":"inference-engine","name":"inference-engine"},{"slug":"infrastructure","name":"infrastructure"},{"slug":"json-api","name":"json-api"},{"slug":"machine-learning","name":"machine-learning"},{"slug":"model-deployment","name":"model-deployment"},{"slug":"php","name":"php"}],"edges":[],"neighbours":[{"slug":"jundot-omlx","name":"omlx","tagline":"LLM inference server with continuous batching and SSD caching for Apple Silicon","github_url":"https://github.com/jundot/omlx","owner":"jundot","repo":"omlx","owner_avatar_url":"https://avatars.githubusercontent.com/u/64250138?v=4","primary_language":"Python","stars":21934,"forks":1899,"topics":["apple-silicon","inference-server","llm","macos","mlx","openai-api"],"archived":false,"github_pushed_at":"2026-09-20T02:23:26+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jundot-omlx","markdown_url":"https://www.graphcanon.com/tools/jundot-omlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jundot-omlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jundot-omlx","shared_categories":["inference-serving"]},{"slug":"triton-inference-server-server","name":"server","tagline":"Optimized cloud and edge inferencing solution","github_url":"https://github.com/triton-inference-server/server","owner":"triton-inference-server","repo":"server","owner_avatar_url":"https://avatars.githubusercontent.com/u/68086070?v=4","primary_language":"Python","stars":10995,"forks":1837,"topics":["cloud","datacenter","deep-learning","edge","gpu","inference","machine-learning"],"archived":false,"github_pushed_at":"2026-09-17T14:35:44+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/triton-inference-server-server","markdown_url":"https://www.graphcanon.com/tools/triton-inference-server-server.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/triton-inference-server-server","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=triton-inference-server-server","shared_categories":["inference-serving"]},{"slug":"runanywhereai-runanywhere-sdks","name":"runanywhere-sdks","tagline":"Production ready toolkit to run AI locally","github_url":"https://github.com/RunanywhereAI/runanywhere-sdks","owner":"RunanywhereAI","repo":"runanywhere-sdks","owner_avatar_url":"https://avatars.githubusercontent.com/u/220821781?v=4","primary_language":"C++","stars":10297,"forks":380,"topics":["android","apple-intelligence","cpp","diffusion-models","edge","flutter","inference","ios","kotlin","llamacpp","llm","multimodal","ollama","on-device-ai","react-native","swift","vlm","voice-ai","web","websdk"],"archived":false,"github_pushed_at":"2026-09-19T18:06:16+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/runanywhereai-runanywhere-sdks","markdown_url":"https://www.graphcanon.com/tools/runanywhereai-runanywhere-sdks.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/runanywhereai-runanywhere-sdks","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=runanywhereai-runanywhere-sdks","shared_categories":["inference-serving"]},{"slug":"bentoml-bentoml","name":"BentoML","tagline":"The easiest way to serve AI apps and models","github_url":"https://github.com/bentoml/BentoML","owner":"bentoml","repo":"BentoML","owner_avatar_url":"https://avatars.githubusercontent.com/u/49176046?v=4","primary_language":"Python","stars":8847,"forks":1032,"topics":["ai-inference","deep-learning","generative-ai","inference-platform","llm","llm-inference","llm-serving","llmops","machine-learning","ml-engineering","mlops","model-inference-service","model-serving","multimodal","python"],"archived":false,"github_pushed_at":"2026-09-07T17:43:01+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/bentoml-bentoml","markdown_url":"https://www.graphcanon.com/tools/bentoml-bentoml.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bentoml-bentoml","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bentoml-bentoml","shared_categories":["inference-serving"]},{"slug":"pytorch-serve","name":"serve","tagline":"Serve, optimize and scale PyTorch models in production","github_url":"https://github.com/pytorch/serve","owner":"pytorch","repo":"serve","owner_avatar_url":"https://avatars.githubusercontent.com/u/21003710?v=4","primary_language":"Java","stars":4344,"forks":880,"topics":["cpu","deep-learning","docker","gpu","kubernetes","machine-learning","metrics","mlops","optimization","pytorch","serving"],"archived":true,"github_pushed_at":"2025-08-06T19:17:08+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/pytorch-serve","markdown_url":"https://www.graphcanon.com/tools/pytorch-serve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pytorch-serve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pytorch-serve","shared_categories":["inference-serving"]},{"slug":"superlinked-sie","name":"sie","tagline":"Open-source inference server and production cluster for all the models your agent needs.","github_url":"https://github.com/superlinked/sie","owner":"superlinked","repo":"sie","owner_avatar_url":"https://avatars.githubusercontent.com/u/94243920?v=4","primary_language":"Python","stars":3293,"forks":311,"topics":["bge","colbert","data-pipeline","deep-learning","embeddings","inference","inference-server","information-retrieval","llm","ml","mlops","natural-language-processing","nlp","python","reranking","retrieval","retrieval-augmented-generation","semantic-search","splade","vector-search"],"archived":false,"github_pushed_at":"2026-09-19T01:00:02+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/superlinked-sie","markdown_url":"https://www.graphcanon.com/tools/superlinked-sie.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/superlinked-sie","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=superlinked-sie","shared_categories":["inference-serving"]},{"slug":"langchain-ai-langserve","name":"langserve","tagline":"LangServe 🦜️🏓","github_url":"https://github.com/langchain-ai/langserve","owner":"langchain-ai","repo":"langserve","owner_avatar_url":"https://avatars.githubusercontent.com/u/126733545?v=4","primary_language":"JavaScript","stars":2328,"forks":270,"topics":["deployment","fastapi","langchain","langchain-python","llm","llms"],"archived":true,"github_pushed_at":"2026-05-05T19:49:07+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/langchain-ai-langserve","markdown_url":"https://www.graphcanon.com/tools/langchain-ai-langserve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langchain-ai-langserve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langchain-ai-langserve","shared_categories":["inference-serving"]},{"slug":"ddalcu-mlx-serve","name":"mlx-serve","tagline":"Native LLM inference server for Apple Silicon","github_url":"https://github.com/ddalcu/mlx-serve","owner":"ddalcu","repo":"mlx-serve","owner_avatar_url":"https://avatars.githubusercontent.com/u/869085?v=4","primary_language":"Zig","stars":1418,"forks":130,"topics":["agent","anthropic-api","apple-silicon","claude-code","deepseek-v4","diffusion","gguf","image-generation","inference","llm","local-llm","macos","macos-app","mlx","openai-api","tool-calling","video-generation","voice-agent","voice-cloning","zig"],"archived":false,"github_pushed_at":"2026-09-19T22:08:35+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/ddalcu-mlx-serve","markdown_url":"https://www.graphcanon.com/tools/ddalcu-mlx-serve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ddalcu-mlx-serve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ddalcu-mlx-serve","shared_categories":["inference-serving"]},{"slug":"basetenlabs-truss","name":"truss","tagline":"The simplest way to serve AI/ML models in production","github_url":"https://github.com/basetenlabs/truss","owner":"basetenlabs","repo":"truss","owner_avatar_url":"https://avatars.githubusercontent.com/u/54861414?v=4","primary_language":"Python","stars":1203,"forks":126,"topics":["artificial-intelligence","easy-to-use","falcon","inference-api","inference-server","machine-learning","model-serving","open-source","packaging","stable-diffusion","whisper","wizardlm"],"archived":false,"github_pushed_at":"2026-09-18T22:12:03+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/basetenlabs-truss","markdown_url":"https://www.graphcanon.com/tools/basetenlabs-truss.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/basetenlabs-truss","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=basetenlabs-truss","shared_categories":["inference-serving"]},{"slug":"underneathall-pinferencia","name":"pinferencia","tagline":"Python library for simplest model inference server","github_url":"https://github.com/underneathall/pinferencia","owner":"underneathall","repo":"pinferencia","owner_avatar_url":"https://avatars.githubusercontent.com/u/76835515?v=4","primary_language":"Python","stars":543,"forks":83,"topics":["ai","artificial-intelligence","computer-vision","data-science","deep-learning","huggingface","inference","inference-server","machine-learning","model-deployment","model-serving","modelserver","nlp","paddlepaddle","predict","python","pytorch","serving","tensorflow","transformers"],"archived":false,"github_pushed_at":"2023-02-14T22:50:48+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/underneathall-pinferencia","markdown_url":"https://www.graphcanon.com/tools/underneathall-pinferencia.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/underneathall-pinferencia","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=underneathall-pinferencia","shared_categories":["inference-serving"]},{"slug":"microsoft-sarathi-serve","name":"sarathi-serve","tagline":"A low-latency and high-throughput serving engine for LLMs","github_url":"https://github.com/microsoft/sarathi-serve","owner":"microsoft","repo":"sarathi-serve","owner_avatar_url":"https://avatars.githubusercontent.com/u/6154722?v=4","primary_language":"Python","stars":527,"forks":67,"topics":["llama","llm-inference","pytorch","transformer"],"archived":false,"github_pushed_at":"2026-01-08T05:10:57+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/microsoft-sarathi-serve","markdown_url":"https://www.graphcanon.com/tools/microsoft-sarathi-serve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/microsoft-sarathi-serve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=microsoft-sarathi-serve","shared_categories":["inference-serving"]},{"slug":"hpcaitech-swiftinfer","name":"SwiftInfer","tagline":"Efficient AI Inference Serving","github_url":"https://github.com/hpcaitech/SwiftInfer","owner":"hpcaitech","repo":"SwiftInfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/88699314?v=4","primary_language":"Python","stars":474,"forks":31,"topics":["artificial-intelligence","deep-learning","gpt","inference","llama","llama2","llm-inference","llm-serving"],"archived":false,"github_pushed_at":"2024-01-08T09:18:42+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/hpcaitech-swiftinfer","markdown_url":"https://www.graphcanon.com/tools/hpcaitech-swiftinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/hpcaitech-swiftinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=hpcaitech-swiftinfer","shared_categories":["inference-serving"]}]}}