{"data":{"node":{"slug":"notai-tech-fastdeploy","name":"fastDeploy","tagline":"Deploy DL/ML inference pipelines with minimal extra code.","github_url":"https://github.com/notAI-tech/fastDeploy","owner":"notAI-tech","repo":"fastDeploy","owner_avatar_url":"https://avatars.githubusercontent.com/u/63401202?v=4","primary_language":"Python","stars":105,"forks":17,"topics":["deep-learning","docker","falcon","gevent","gunicorn","http-server","inference-server","model-deployment","model-serving","python","pytorch","serving","streaming-audio","tensorflow-serving","tf-serving","torchserve","triton","triton-inference-server","triton-server","websocket"],"archived":false,"github_pushed_at":"2026-02-10T16:18:52+00:00","maintenance_label":"Slowing","stars_delta_30d":0,"url":"https://www.graphcanon.com/tools/notai-tech-fastdeploy","markdown_url":"https://www.graphcanon.com/tools/notai-tech-fastdeploy.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/notai-tech-fastdeploy","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=notai-tech-fastdeploy"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"deep-learning","name":"deep-learning"},{"slug":"docker","name":"docker"},{"slug":"falcon","name":"falcon"},{"slug":"gevent","name":"gevent"},{"slug":"gunicorn","name":"gunicorn"},{"slug":"http-server","name":"http-server"},{"slug":"model-serving","name":"model-serving"},{"slug":"python","name":"python"}],"edges":[],"neighbours":[{"slug":"mozilla-ai-llamafile","name":"llamafile","tagline":"Distribute and run LLMs with a single file.","github_url":"https://github.com/mozilla-ai/llamafile","owner":"mozilla-ai","repo":"llamafile","owner_avatar_url":"https://avatars.githubusercontent.com/u/129804596?v=4","primary_language":"C++","stars":25999,"forks":1598,"topics":["cross-platform","gguf","llama-cpp","local-ai","local-inference","local-llm","open-source-ai","single-file-executable","speech-to-text"],"archived":false,"github_pushed_at":"2026-09-16T09:14:28+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/mozilla-ai-llamafile","markdown_url":"https://www.graphcanon.com/tools/mozilla-ai-llamafile.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/mozilla-ai-llamafile","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=mozilla-ai-llamafile","shared_categories":["inference-serving"]},{"slug":"runanywhereai-runanywhere-sdks","name":"runanywhere-sdks","tagline":"Production ready toolkit to run AI locally","github_url":"https://github.com/RunanywhereAI/runanywhere-sdks","owner":"RunanywhereAI","repo":"runanywhere-sdks","owner_avatar_url":"https://avatars.githubusercontent.com/u/220821781?v=4","primary_language":"C++","stars":10297,"forks":380,"topics":["android","apple-intelligence","cpp","diffusion-models","edge","flutter","inference","ios","kotlin","llamacpp","llm","multimodal","ollama","on-device-ai","react-native","swift","vlm","voice-ai","web","websdk"],"archived":false,"github_pushed_at":"2026-09-19T18:06:16+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/runanywhereai-runanywhere-sdks","markdown_url":"https://www.graphcanon.com/tools/runanywhereai-runanywhere-sdks.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/runanywhereai-runanywhere-sdks","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=runanywhereai-runanywhere-sdks","shared_categories":["inference-serving"]},{"slug":"huggingface-accelerate","name":"accelerate","tagline":"A tool for launching, training, and using PyTorch models with ease on various devices, configurations, including mixed precision support.","github_url":"https://github.com/huggingface/accelerate","owner":"huggingface","repo":"accelerate","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":9841,"forks":1462,"topics":[],"archived":false,"github_pushed_at":"2026-09-02T17:09:00+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-accelerate","markdown_url":"https://www.graphcanon.com/tools/huggingface-accelerate.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-accelerate","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-accelerate","shared_categories":["inference-serving"]},{"slug":"bentoml-bentoml","name":"BentoML","tagline":"The easiest way to serve AI apps and models","github_url":"https://github.com/bentoml/BentoML","owner":"bentoml","repo":"BentoML","owner_avatar_url":"https://avatars.githubusercontent.com/u/49176046?v=4","primary_language":"Python","stars":8847,"forks":1032,"topics":["ai-inference","deep-learning","generative-ai","inference-platform","llm","llm-inference","llm-serving","llmops","machine-learning","ml-engineering","mlops","model-inference-service","model-serving","multimodal","python"],"archived":false,"github_pushed_at":"2026-09-07T17:43:01+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/bentoml-bentoml","markdown_url":"https://www.graphcanon.com/tools/bentoml-bentoml.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bentoml-bentoml","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bentoml-bentoml","shared_categories":["inference-serving"]},{"slug":"kserve-kserve","name":"kserve","tagline":"Standardized Distributed Generative and Predictive AI Inference Platform for Scalable, Multi-Framework Deployment on Kubernetes","github_url":"https://github.com/kserve/kserve","owner":"kserve","repo":"kserve","owner_avatar_url":"https://avatars.githubusercontent.com/u/83512434?v=4","primary_language":"Go","stars":5970,"forks":1689,"topics":["artificial-intelligence","cncf","genai","hacktoberfest","istio","k8s","knative","kserve","kubeflow","kubernetes","llm-inference","machine-learning","mlops","model-interpretability","model-serving","pytorch","service-mesh","tensorflow","vllm","xgboost"],"archived":false,"github_pushed_at":"2026-09-19T13:08:33+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/kserve-kserve","markdown_url":"https://www.graphcanon.com/tools/kserve-kserve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kserve-kserve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kserve-kserve","shared_categories":["inference-serving"]},{"slug":"seldonio-seldon-core","name":"seldon-core","tagline":"An MLOps framework to package, deploy, monitor and manage thousands of production machine learning models","github_url":"https://github.com/SeldonIO/seldon-core","owner":"SeldonIO","repo":"seldon-core","owner_avatar_url":"https://avatars.githubusercontent.com/u/10297834?v=4","primary_language":"Go","stars":4780,"forks":868,"topics":["aiops","deployment","kubernetes","machine-learning","machine-learning-operations","mlops","production-machine-learning","serving"],"archived":false,"github_pushed_at":"2026-03-23T11:39:54+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/seldonio-seldon-core","markdown_url":"https://www.graphcanon.com/tools/seldonio-seldon-core.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/seldonio-seldon-core","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=seldonio-seldon-core","shared_categories":["inference-serving"]},{"slug":"pytorch-serve","name":"serve","tagline":"Serve, optimize and scale PyTorch models in production","github_url":"https://github.com/pytorch/serve","owner":"pytorch","repo":"serve","owner_avatar_url":"https://avatars.githubusercontent.com/u/21003710?v=4","primary_language":"Java","stars":4344,"forks":880,"topics":["cpu","deep-learning","docker","gpu","kubernetes","machine-learning","metrics","mlops","optimization","pytorch","serving"],"archived":true,"github_pushed_at":"2025-08-06T19:17:08+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/pytorch-serve","markdown_url":"https://www.graphcanon.com/tools/pytorch-serve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pytorch-serve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pytorch-serve","shared_categories":["inference-serving"]},{"slug":"kubeflow-pipelines","name":"pipelines","tagline":"Machine Learning Pipelines for Kubeflow","github_url":"https://github.com/kubeflow/pipelines","owner":"kubeflow","repo":"pipelines","owner_avatar_url":"https://avatars.githubusercontent.com/u/33164907?v=4","primary_language":"Python","stars":4204,"forks":2106,"topics":["data-science","kubeflow","kubeflow-pipelines","kubernetes","machine-learning","mlops","pipeline"],"archived":false,"github_pushed_at":"2026-09-02T18:27:06+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/kubeflow-pipelines","markdown_url":"https://www.graphcanon.com/tools/kubeflow-pipelines.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kubeflow-pipelines","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kubeflow-pipelines","shared_categories":["inference-serving"]},{"slug":"superlinked-sie","name":"sie","tagline":"Open-source inference server and production cluster for all the models your agent needs.","github_url":"https://github.com/superlinked/sie","owner":"superlinked","repo":"sie","owner_avatar_url":"https://avatars.githubusercontent.com/u/94243920?v=4","primary_language":"Python","stars":3293,"forks":311,"topics":["bge","colbert","data-pipeline","deep-learning","embeddings","inference","inference-server","information-retrieval","llm","ml","mlops","natural-language-processing","nlp","python","reranking","retrieval","retrieval-augmented-generation","semantic-search","splade","vector-search"],"archived":false,"github_pushed_at":"2026-09-19T01:00:02+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/superlinked-sie","markdown_url":"https://www.graphcanon.com/tools/superlinked-sie.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/superlinked-sie","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=superlinked-sie","shared_categories":["inference-serving"]},{"slug":"b4rtaz-distributed-llama","name":"distributed-llama","tagline":"Distributed LLM inference using home devices cluster","github_url":"https://github.com/b4rtaz/distributed-llama","owner":"b4rtaz","repo":"distributed-llama","owner_avatar_url":"https://avatars.githubusercontent.com/u/12797776?v=4","primary_language":"C++","stars":3060,"forks":250,"topics":["distributed-computing","distributed-llm","llama2","llama3","llm","llm-inference","llms","neural-network","open-llm"],"archived":false,"github_pushed_at":"2026-07-05T16:47:20+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama","markdown_url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/b4rtaz-distributed-llama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=b4rtaz-distributed-llama","shared_categories":["inference-serving"]},{"slug":"jina-ai-langchain-serve","name":"langchain-serve","tagline":"Self-host LLM Apps with Docker Compose or Kubernetes","github_url":"https://github.com/jina-ai/langchain-serve","owner":"jina-ai","repo":"langchain-serve","owner_avatar_url":"https://avatars.githubusercontent.com/u/60539444?v=4","primary_language":"Python","stars":1639,"forks":133,"topics":["autogpt","autonomous-agents","babyagi","chatbot","fastapi","gpt","langchain","llm","production","python","slack"],"archived":true,"github_pushed_at":"2023-09-20T04:01:50+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/jina-ai-langchain-serve","markdown_url":"https://www.graphcanon.com/tools/jina-ai-langchain-serve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jina-ai-langchain-serve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jina-ai-langchain-serve","shared_categories":["inference-serving"]},{"slug":"tensorflow-model-optimization","name":"model-optimization","tagline":"Toolkit for optimizing ML models in Keras and TensorFlow","github_url":"https://github.com/tensorflow/model-optimization","owner":"tensorflow","repo":"model-optimization","owner_avatar_url":"https://avatars.githubusercontent.com/u/15658638?v=4","primary_language":"Python","stars":1579,"forks":346,"topics":["compression","deep-learning","keras","machine-learning","ml","model-compression","optimization","pruning","quantization","quantized-networks","quantized-neural-networks","quantized-training","sparsity","tensorflow"],"archived":false,"github_pushed_at":"2026-08-24T15:24:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/tensorflow-model-optimization","markdown_url":"https://www.graphcanon.com/tools/tensorflow-model-optimization.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tensorflow-model-optimization","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tensorflow-model-optimization","shared_categories":[]}]}}