{"data":{"node":{"slug":"beam-cloud-beta9","name":"beta9","tagline":"Ultrafast serverless GPU inference, sandboxes, and background jobs","github_url":"https://github.com/beam-cloud/beta9","owner":"beam-cloud","repo":"beta9","owner_avatar_url":"https://avatars.githubusercontent.com/u/126977053?v=4","primary_language":"Go","stars":1753,"forks":158,"topics":["autoscaler","cloudrun","cuda","developer-productivity","distributed-computing","faas","fine-tuning","functions-as-a-service","generative-ai","gpu","large-language-models","llm","llm-inference","ml-platform","paas","self-hosted","serverless","serverless-containers"],"archived":false,"github_pushed_at":"2026-08-19T19:00:48+00:00","maintenance_label":"Very active","stars_delta_30d":33,"url":"https://www.graphcanon.com/tools/beam-cloud-beta9","markdown_url":"https://www.graphcanon.com/tools/beam-cloud-beta9.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/beam-cloud-beta9","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=beam-cloud-beta9"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"},{"slug":"llm-frameworks","name":"LLM Frameworks","url":"https://www.graphcanon.com/categories/llm-frameworks","markdown_url":"https://www.graphcanon.com/categories/llm-frameworks.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/llm-frameworks"}],"tags":[{"slug":"autoscaler","name":"autoscaler"},{"slug":"cloudrun","name":"cloudrun"},{"slug":"cuda","name":"cuda"},{"slug":"distributed-computing","name":"distributed-computing"},{"slug":"faas","name":"faas"},{"slug":"fine-tuning","name":"fine-tuning"},{"slug":"functions-as-a-service","name":"functions-as-a-service"},{"slug":"generative-ai","name":"generative-ai"}],"edges":[],"neighbours":[{"slug":"skypilot-org-skypilot","name":"skypilot","tagline":"Run, manage, and scale AI workloads on any AI infrastructure.","github_url":"https://github.com/skypilot-org/skypilot","owner":"skypilot-org","repo":"skypilot","owner_avatar_url":"https://avatars.githubusercontent.com/u/109387420?v=4","primary_language":"Python","stars":10456,"forks":1175,"topics":["cloud-computing","cloud-management","cost-optimization","deep-learning","distributed-training","gpu","hyperparameter-tuning","job-queue","job-scheduler","llm-serving","llm-training","machine-learning","ml-infrastructure","ml-platform","mlops","multicloud","slurm","spot-instances","tpu"],"archived":false,"github_pushed_at":"2026-08-07T10:25:54+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/skypilot-org-skypilot","markdown_url":"https://www.graphcanon.com/tools/skypilot-org-skypilot.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/skypilot-org-skypilot","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=skypilot-org-skypilot","shared_categories":["inference-serving"]},{"slug":"tensorflow-serving","name":"serving","tagline":"A flexible, high-performance serving system for machine learning models","github_url":"https://github.com/tensorflow/serving","owner":"tensorflow","repo":"serving","owner_avatar_url":"https://avatars.githubusercontent.com/u/15658638?v=4","primary_language":"C++","stars":6359,"forks":2204,"topics":["cpp","deep-learning","deep-neural-networks","machine-learning","ml","neural-network","python","serving","tensorflow"],"archived":false,"github_pushed_at":"2026-07-30T07:02:43+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/tensorflow-serving","markdown_url":"https://www.graphcanon.com/tools/tensorflow-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tensorflow-serving","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tensorflow-serving","shared_categories":["inference-serving"]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"tensorchord-awesome-llmops","name":"Awesome-LLMOps","tagline":"An awesome & curated list of best LLMOps tools for developers","github_url":"https://github.com/tensorchord/Awesome-LLMOps","owner":"tensorchord","repo":"Awesome-LLMOps","owner_avatar_url":"https://avatars.githubusercontent.com/u/100543303?v=4","primary_language":"Shell","stars":5915,"forks":993,"topics":["ai-development-tools","awesome-list","llmops","mlops"],"archived":false,"github_pushed_at":"2026-05-21T09:12:50+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/tensorchord-awesome-llmops","markdown_url":"https://www.graphcanon.com/tools/tensorchord-awesome-llmops.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tensorchord-awesome-llmops","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tensorchord-awesome-llmops","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"kserve-kserve","name":"kserve","tagline":"Standardized Distributed Generative and Predictive AI Inference Platform for Scalable, Multi-Framework Deployment on Kubernetes","github_url":"https://github.com/kserve/kserve","owner":"kserve","repo":"kserve","owner_avatar_url":"https://avatars.githubusercontent.com/u/83512434?v=4","primary_language":"Go","stars":5826,"forks":1632,"topics":["artificial-intelligence","cncf","genai","hacktoberfest","istio","k8s","knative","kserve","kubeflow","kubernetes","llm-inference","machine-learning","mlops","model-interpretability","model-serving","pytorch","service-mesh","tensorflow","vllm","xgboost"],"archived":false,"github_pushed_at":"2026-08-24T02:37:23+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/kserve-kserve","markdown_url":"https://www.graphcanon.com/tools/kserve-kserve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kserve-kserve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kserve-kserve","shared_categories":["inference-serving"]},{"slug":"gpustack-gpustack","name":"gpustack","tagline":"A GPU cluster manager for high-performance AI model serving and on-demand SSH-accessible GPU instances","github_url":"https://github.com/gpustack/gpustack","owner":"gpustack","repo":"gpustack","owner_avatar_url":"https://avatars.githubusercontent.com/u/169020824?v=4","primary_language":"Python","stars":5454,"forks":609,"topics":["ascend","cuda","deepseek","distributed-inference","genai","high-performance-inference","inference","llama","llm","llm-inference","llm-serving","maas","mindie","openai","qwen","rocm","sglang","vllm"],"archived":false,"github_pushed_at":"2026-08-07T03:09:44+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/gpustack-gpustack","markdown_url":"https://www.graphcanon.com/tools/gpustack-gpustack.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/gpustack-gpustack","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=gpustack-gpustack","shared_categories":["inference-serving"]},{"slug":"huaizhengzhang-ai-infra-from-zero-to-hero","name":"AI-Infra-from-Zero-to-Hero","tagline":"Awesome System for Machine Learning and LLM Infra","github_url":"https://github.com/HuaizhengZhang/AI-Infra-from-Zero-to-Hero","owner":"HuaizhengZhang","repo":"AI-Infra-from-Zero-to-Hero","owner_avatar_url":"https://avatars.githubusercontent.com/u/5894780?v=4","primary_language":null,"stars":4285,"forks":409,"topics":["ai-infra","genai","large-language-models","llmsys","mlsys","model-serving","model-training"],"archived":false,"github_pushed_at":"2025-07-25T02:24:35+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/huaizhengzhang-ai-infra-from-zero-to-hero","markdown_url":"https://www.graphcanon.com/tools/huaizhengzhang-ai-infra-from-zero-to-hero.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huaizhengzhang-ai-infra-from-zero-to-hero","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huaizhengzhang-ai-infra-from-zero-to-hero","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"a16z-infra-ai-getting-started","name":"ai-getting-started","tagline":"A Javascript AI getting started stack for weekend projects","github_url":"https://github.com/a16z-infra/ai-getting-started","owner":"a16z-infra","repo":"ai-getting-started","owner_avatar_url":"https://avatars.githubusercontent.com/u/130202746?v=4","primary_language":"TypeScript","stars":4141,"forks":660,"topics":[],"archived":false,"github_pushed_at":"2024-08-21T12:35:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/a16z-infra-ai-getting-started","markdown_url":"https://www.graphcanon.com/tools/a16z-infra-ai-getting-started.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/a16z-infra-ai-getting-started","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=a16z-infra-ai-getting-started","shared_categories":[]},{"slug":"b4rtaz-distributed-llama","name":"distributed-llama","tagline":"Distributed LLM inference using home devices cluster","github_url":"https://github.com/b4rtaz/distributed-llama","owner":"b4rtaz","repo":"distributed-llama","owner_avatar_url":"https://avatars.githubusercontent.com/u/12797776?v=4","primary_language":"C++","stars":3044,"forks":246,"topics":["distributed-computing","distributed-llm","llama2","llama3","llm","llm-inference","llms","neural-network","open-llm"],"archived":false,"github_pushed_at":"2026-07-05T16:47:20+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama","markdown_url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/b4rtaz-distributed-llama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=b4rtaz-distributed-llama","shared_categories":["inference-serving"]},{"slug":"michaelfeil-infinity","name":"infinity","tagline":"High-throughput, low-latency serving engine for text-embeddings and various models","github_url":"https://github.com/michaelfeil/infinity","owner":"michaelfeil","repo":"infinity","owner_avatar_url":"https://avatars.githubusercontent.com/u/63565275?v=4","primary_language":"Python","stars":2907,"forks":196,"topics":["bert-embeddings","llm","text-embeddings"],"archived":false,"github_pushed_at":"2026-03-24T03:59:47+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/michaelfeil-infinity","markdown_url":"https://www.graphcanon.com/tools/michaelfeil-infinity.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/michaelfeil-infinity","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=michaelfeil-infinity","shared_categories":["inference-serving"]},{"slug":"superlinked-sie","name":"sie","tagline":"Open-source inference server and production cluster for all the models your agent needs.","github_url":"https://github.com/superlinked/sie","owner":"superlinked","repo":"sie","owner_avatar_url":"https://avatars.githubusercontent.com/u/94243920?v=4","primary_language":"Python","stars":2804,"forks":272,"topics":["bge","colbert","data-pipeline","deep-learning","embeddings","inference","inference-server","information-retrieval","llm","ml","mlops","natural-language-processing","nlp","python","reranking","retrieval","retrieval-augmented-generation","semantic-search","splade","vector-search"],"archived":false,"github_pushed_at":"2026-08-21T20:28:04+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/superlinked-sie","markdown_url":"https://www.graphcanon.com/tools/superlinked-sie.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/superlinked-sie","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=superlinked-sie","shared_categories":["inference-serving"]},{"slug":"microsoft-pai","name":"pai","tagline":"Resource scheduling and cluster management for AI","github_url":"https://github.com/microsoft/pai","owner":"microsoft","repo":"pai","owner_avatar_url":"https://avatars.githubusercontent.com/u/6154722?v=4","primary_language":"JavaScript","stars":2686,"forks":549,"topics":["ai","artificial-intelligence","chainer","cloud","cluster-management","cluster-manager","gpu","gpu-cluster","gpu-computing","gpu-scheduler","jupyter","kubernetes","machine-learning","model-training","on-premise","pytorch","resource-management","scheduling","tensorflow"],"archived":true,"github_pushed_at":"2024-06-06T07:56:07+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/microsoft-pai","markdown_url":"https://www.graphcanon.com/tools/microsoft-pai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/microsoft-pai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=microsoft-pai","shared_categories":["inference-serving"]}]}}