{"data":{"node":{"slug":"xllm-ai-xllm","name":"xllm","tagline":"A high-performance inference engine for LLM, VLM, DiT and REC models","github_url":"https://github.com/xLLM-AI/xllm","owner":"xLLM-AI","repo":"xllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/205719415?v=4","primary_language":"C++","stars":1534,"forks":282,"topics":["deepseek","glm","inference","inference-engine","large-language-models","llm-inference","qwen"],"archived":false,"github_pushed_at":"2026-08-24T09:51:10+00:00","maintenance_label":"Very active","stars_delta_30d":41,"url":"https://www.graphcanon.com/tools/xllm-ai-xllm","markdown_url":"https://www.graphcanon.com/tools/xllm-ai-xllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xllm-ai-xllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xllm-ai-xllm"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"deepseek","name":"deepseek"},{"slug":"glm","name":"glm"},{"slug":"llm-inference","name":"llm-inference"}],"edges":[],"neighbours":[{"slug":"vllm-project-vllm","name":"vllm","tagline":"A high-throughput and memory-efficient inference and serving engine for LLMs","github_url":"https://github.com/vllm-project/vllm","owner":"vllm-project","repo":"vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/136984999?v=4","primary_language":"Python","stars":87847,"forks":20135,"topics":["amd","blackwell","cuda","deepseek","deepseek-v3","gpt","gpt-oss","inference","kimi","llama","llm","llm-serving","model-serving","moe","openai","pytorch","qwen","qwen3","tpu","transformer"],"archived":false,"github_pushed_at":"2026-08-01T11:55:36+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/vllm-project-vllm","markdown_url":"https://www.graphcanon.com/tools/vllm-project-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vllm-project-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vllm-project-vllm","shared_categories":["inference-serving"]},{"slug":"nomic-ai-gpt4all","name":"gpt4all","tagline":"Run Local LLMs on Any Device","github_url":"https://github.com/nomic-ai/gpt4all","owner":"nomic-ai","repo":"gpt4all","owner_avatar_url":"https://avatars.githubusercontent.com/u/102670180?v=4","primary_language":"C++","stars":77393,"forks":8296,"topics":["ai-chat","llm-inference"],"archived":false,"github_pushed_at":"2025-05-27T20:05:19+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/nomic-ai-gpt4all","markdown_url":"https://www.graphcanon.com/tools/nomic-ai-gpt4all.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nomic-ai-gpt4all","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nomic-ai-gpt4all","shared_categories":["inference-serving"]},{"slug":"alexsjones-llmfit","name":"llmfit","tagline":"Hundreds of models & providers. One command to find what runs on your hardware.","github_url":"https://github.com/AlexsJones/llmfit","owner":"AlexsJones","repo":"llmfit","owner_avatar_url":"https://avatars.githubusercontent.com/u/1235925?v=4","primary_language":"Rust","stars":31867,"forks":1978,"topics":["gguf","llm","localai","mlx","skill","unsloth"],"archived":false,"github_pushed_at":"2026-08-14T07:36:41+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/alexsjones-llmfit","markdown_url":"https://www.graphcanon.com/tools/alexsjones-llmfit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alexsjones-llmfit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alexsjones-llmfit","shared_categories":[]},{"slug":"jundot-omlx","name":"omlx","tagline":"LLM inference server with continuous batching and SSD caching for Apple Silicon","github_url":"https://github.com/jundot/omlx","owner":"jundot","repo":"omlx","owner_avatar_url":"https://avatars.githubusercontent.com/u/64250138?v=4","primary_language":"Python","stars":18679,"forks":1617,"topics":["apple-silicon","inference-server","llm","macos","mlx","openai-api"],"archived":false,"github_pushed_at":"2026-08-14T09:12:48+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/jundot-omlx","markdown_url":"https://www.graphcanon.com/tools/jundot-omlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jundot-omlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jundot-omlx","shared_categories":["inference-serving"]},{"slug":"lightning-ai-litgpt","name":"litgpt","tagline":"High-performance LLMs with recipes for pretraining, finetuning and deployment","github_url":"https://github.com/Lightning-AI/litgpt","owner":"Lightning-AI","repo":"litgpt","owner_avatar_url":"https://avatars.githubusercontent.com/u/58386951?v=4","primary_language":"Python","stars":13605,"forks":1483,"topics":["ai","artificial-intelligence","deep-learning","large-language-models","llm","llm-inference","llms"],"archived":false,"github_pushed_at":"2026-07-20T10:24:12+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lightning-ai-litgpt","markdown_url":"https://www.graphcanon.com/tools/lightning-ai-litgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lightning-ai-litgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lightning-ai-litgpt","shared_categories":["inference-serving"]},{"slug":"eugeneyan-open-llms","name":"open-llms","tagline":"A list of open LLMs available for commercial use.","github_url":"https://github.com/eugeneyan/open-llms","owner":"eugeneyan","repo":"open-llms","owner_avatar_url":"https://avatars.githubusercontent.com/u/6831355?v=4","primary_language":null,"stars":12849,"forks":985,"topics":["commercial","large-language-models","llm","llms"],"archived":false,"github_pushed_at":"2025-02-13T06:37:12+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/eugeneyan-open-llms","markdown_url":"https://www.graphcanon.com/tools/eugeneyan-open-llms.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eugeneyan-open-llms","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eugeneyan-open-llms","shared_categories":[]},{"slug":"xorbitsai-inference","name":"inference","tagline":"Unified production-ready inference API for various models","github_url":"https://github.com/xorbitsai/inference","owner":"xorbitsai","repo":"inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/109655068?v=4","primary_language":"Python","stars":9470,"forks":851,"topics":["artificial-intelligence","chatglm","deployment","flan-t5","gemma","ggml","glm4","inference","llama","llama3","llamacpp","llm","machine-learning","mistral","openai-api","pytorch","qwen","vllm","whisper","wizardlm"],"archived":false,"github_pushed_at":"2026-08-02T05:44:21+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/xorbitsai-inference","markdown_url":"https://www.graphcanon.com/tools/xorbitsai-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xorbitsai-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xorbitsai-inference","shared_categories":["inference-serving"]},{"slug":"andyyyy64-whichllm","name":"whichllm","tagline":"Command-line tool to find and benchmark local LLM performance","github_url":"https://github.com/Andyyyy64/whichllm","owner":"Andyyyy64","repo":"whichllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/105579829?v=4","primary_language":"Python","stars":6225,"forks":330,"topics":["ai","apple-silicon","benchmarks","cli","command-line-tool","gguf","gpu","huggingface","inference","llm","local-llm","ollama","python","vram"],"archived":false,"github_pushed_at":"2026-08-05T07:15:32+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/andyyyy64-whichllm","markdown_url":"https://www.graphcanon.com/tools/andyyyy64-whichllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/andyyyy64-whichllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=andyyyy64-whichllm","shared_categories":["inference-serving"]},{"slug":"raullenchai-rapid-mlx","name":"Rapid-MLX","tagline":"Fast local AI engine for Apple Silicon","github_url":"https://github.com/raullenchai/Rapid-MLX","owner":"raullenchai","repo":"Rapid-MLX","owner_avatar_url":"https://avatars.githubusercontent.com/u/989846?v=4","primary_language":"Python","stars":3391,"forks":388,"topics":["apple-silicon","claude-code","cursor","deepseek","fastapi","hacktoberfest","inference","llm","local-llm","m1","m2","m3","macos","mlx","ollama-alternative","openai-api","python","qwen","tool-calling"],"archived":false,"github_pushed_at":"2026-08-01T23:14:51+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/raullenchai-rapid-mlx","markdown_url":"https://www.graphcanon.com/tools/raullenchai-rapid-mlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/raullenchai-rapid-mlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=raullenchai-rapid-mlx","shared_categories":["inference-serving"]},{"slug":"b4rtaz-distributed-llama","name":"distributed-llama","tagline":"Distributed LLM inference using home devices cluster","github_url":"https://github.com/b4rtaz/distributed-llama","owner":"b4rtaz","repo":"distributed-llama","owner_avatar_url":"https://avatars.githubusercontent.com/u/12797776?v=4","primary_language":"C++","stars":3044,"forks":246,"topics":["distributed-computing","distributed-llm","llama2","llama3","llm","llm-inference","llms","neural-network","open-llm"],"archived":false,"github_pushed_at":"2026-07-05T16:47:20+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama","markdown_url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/b4rtaz-distributed-llama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=b4rtaz-distributed-llama","shared_categories":["inference-serving"]},{"slug":"turboderp-exllama","name":"exllama","tagline":"Memory-efficient rewrite of HF transformers for Llama with quantized weights","github_url":"https://github.com/turboderp/exllama","owner":"turboderp","repo":"exllama","owner_avatar_url":"https://avatars.githubusercontent.com/u/11859846?v=4","primary_language":"Python","stars":2937,"forks":220,"topics":[],"archived":false,"github_pushed_at":"2023-09-30T19:06:04+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/turboderp-exllama","markdown_url":"https://www.graphcanon.com/tools/turboderp-exllama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/turboderp-exllama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=turboderp-exllama","shared_categories":["inference-serving"]},{"slug":"stochasticai-xturing","name":"xTuring","tagline":"Personalize and control open-source LLMs with ease","github_url":"https://github.com/stochasticai/xTuring","owner":"stochasticai","repo":"xTuring","owner_avatar_url":"https://avatars.githubusercontent.com/u/66399337?v=4","primary_language":"Python","stars":2674,"forks":211,"topics":["adapter","deep-learning","fine-tuning","finetuning","gen-ai","generative-ai","gpt-2","gpt-j","language-model","llama","llm","lora","mistral","mixed-precision","peft","quantization"],"archived":false,"github_pushed_at":"2026-03-04T23:07:06+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/stochasticai-xturing","markdown_url":"https://www.graphcanon.com/tools/stochasticai-xturing.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/stochasticai-xturing","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=stochasticai-xturing","shared_categories":[]}]}}