{"data":{"node":{"slug":"vllm-project-vllm-ascend","name":"vllm-ascend","tagline":"Community maintained hardware plugin for vLLM on Ascend","github_url":"https://github.com/vllm-project/vllm-ascend","owner":"vllm-project","repo":"vllm-ascend","owner_avatar_url":"https://avatars.githubusercontent.com/u/136984999?v=4","primary_language":"C++","stars":2674,"forks":2081,"topics":["ascend","inference","llm","llm-serving","llmops","mlops","model-serving","transformer","vllm"],"archived":false,"github_pushed_at":"2026-08-20T12:01:45+00:00","maintenance_label":"Very active","stars_delta_30d":230,"url":"https://www.graphcanon.com/tools/vllm-project-vllm-ascend","markdown_url":"https://www.graphcanon.com/tools/vllm-project-vllm-ascend.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vllm-project-vllm-ascend","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vllm-project-vllm-ascend"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"ascend","name":"ascend"},{"slug":"inference","name":"inference"},{"slug":"llm","name":"llm"},{"slug":"llm-serving","name":"llm-serving"},{"slug":"model-serving","name":"model-serving"}],"edges":[],"neighbours":[{"slug":"vllm-project-vllm","name":"vllm","tagline":"A high-throughput and memory-efficient inference and serving engine for LLMs","github_url":"https://github.com/vllm-project/vllm","owner":"vllm-project","repo":"vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/136984999?v=4","primary_language":"Python","stars":87847,"forks":20135,"topics":["amd","blackwell","cuda","deepseek","deepseek-v3","gpt","gpt-oss","inference","kimi","llama","llm","llm-serving","model-serving","moe","openai","pytorch","qwen","qwen3","tpu","transformer"],"archived":false,"github_pushed_at":"2026-08-01T11:55:36+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/vllm-project-vllm","markdown_url":"https://www.graphcanon.com/tools/vllm-project-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vllm-project-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vllm-project-vllm","shared_categories":["inference-serving"]},{"slug":"alexsjones-llmfit","name":"llmfit","tagline":"Hundreds of models & providers. One command to find what runs on your hardware.","github_url":"https://github.com/AlexsJones/llmfit","owner":"AlexsJones","repo":"llmfit","owner_avatar_url":"https://avatars.githubusercontent.com/u/1235925?v=4","primary_language":"Rust","stars":31867,"forks":1978,"topics":["gguf","llm","localai","mlx","skill","unsloth"],"archived":false,"github_pushed_at":"2026-08-14T07:36:41+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/alexsjones-llmfit","markdown_url":"https://www.graphcanon.com/tools/alexsjones-llmfit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alexsjones-llmfit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alexsjones-llmfit","shared_categories":[]},{"slug":"jundot-omlx","name":"omlx","tagline":"LLM inference server with continuous batching and SSD caching for Apple Silicon","github_url":"https://github.com/jundot/omlx","owner":"jundot","repo":"omlx","owner_avatar_url":"https://avatars.githubusercontent.com/u/64250138?v=4","primary_language":"Python","stars":18679,"forks":1617,"topics":["apple-silicon","inference-server","llm","macos","mlx","openai-api"],"archived":false,"github_pushed_at":"2026-08-14T09:12:48+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/jundot-omlx","markdown_url":"https://www.graphcanon.com/tools/jundot-omlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jundot-omlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jundot-omlx","shared_categories":["inference-serving"]},{"slug":"lightning-ai-litgpt","name":"litgpt","tagline":"High-performance LLMs with recipes for pretraining, finetuning and deployment","github_url":"https://github.com/Lightning-AI/litgpt","owner":"Lightning-AI","repo":"litgpt","owner_avatar_url":"https://avatars.githubusercontent.com/u/58386951?v=4","primary_language":"Python","stars":13605,"forks":1483,"topics":["ai","artificial-intelligence","deep-learning","large-language-models","llm","llm-inference","llms"],"archived":false,"github_pushed_at":"2026-07-20T10:24:12+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lightning-ai-litgpt","markdown_url":"https://www.graphcanon.com/tools/lightning-ai-litgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lightning-ai-litgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lightning-ai-litgpt","shared_categories":["inference-serving"]},{"slug":"b4rtaz-distributed-llama","name":"distributed-llama","tagline":"Distributed LLM inference using home devices cluster","github_url":"https://github.com/b4rtaz/distributed-llama","owner":"b4rtaz","repo":"distributed-llama","owner_avatar_url":"https://avatars.githubusercontent.com/u/12797776?v=4","primary_language":"C++","stars":3012,"forks":242,"topics":["distributed-computing","distributed-llm","llama2","llama3","llm","llm-inference","llms","neural-network","open-llm"],"archived":false,"github_pushed_at":"2026-07-05T16:47:20+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama","markdown_url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/b4rtaz-distributed-llama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=b4rtaz-distributed-llama","shared_categories":["inference-serving"]},{"slug":"rafska-awesome-local-llm","name":"awesome-local-llm","tagline":"Resources for running LLMs locally","github_url":"https://github.com/rafska/awesome-local-llm","owner":"rafska","repo":"awesome-local-llm","owner_avatar_url":"https://avatars.githubusercontent.com/u/17859377?v=4","primary_language":null,"stars":2518,"forks":316,"topics":["ai","awesome","awesome-list","llm","local","local-ai","local-llm","resources","self-hosted","selfhosted"],"archived":false,"github_pushed_at":"2026-08-04T23:08:15+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/rafska-awesome-local-llm","markdown_url":"https://www.graphcanon.com/tools/rafska-awesome-local-llm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rafska-awesome-local-llm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rafska-awesome-local-llm","shared_categories":["inference-serving"]},{"slug":"xllm-ai-xllm","name":"xllm","tagline":"A high-performance inference engine for LLM, VLM, DiT and REC models","github_url":"https://github.com/xLLM-AI/xllm","owner":"xLLM-AI","repo":"xllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/205719415?v=4","primary_language":"C++","stars":1493,"forks":269,"topics":["deepseek","glm","inference","inference-engine","large-language-models","llm-inference","qwen"],"archived":false,"github_pushed_at":"2026-07-24T10:38:15+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/xllm-ai-xllm","markdown_url":"https://www.graphcanon.com/tools/xllm-ai-xllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xllm-ai-xllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xllm-ai-xllm","shared_categories":["inference-serving"]},{"slug":"waybarrios-vllm-mlx","name":"vllm-mlx","tagline":"Server for LLMs and vision-language models compatible with Apple Silicon","github_url":"https://github.com/waybarrios/vllm-mlx","owner":"waybarrios","repo":"vllm-mlx","owner_avatar_url":"https://avatars.githubusercontent.com/u/6794828?v=4","primary_language":"Python","stars":1472,"forks":205,"topics":["anthropic","apple-silicon","audio-processing","claude-code","computer-vision","image-understanding","inference","llm","machine-learning","macos","mllm","mlx","multimodal-ai","speech-to-text","stt","text-to-speech","tts","video-understanding","vision-language-model","vllm"],"archived":false,"github_pushed_at":"2026-06-28T20:18:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/waybarrios-vllm-mlx","markdown_url":"https://www.graphcanon.com/tools/waybarrios-vllm-mlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/waybarrios-vllm-mlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=waybarrios-vllm-mlx","shared_categories":["inference-serving"]},{"slug":"jmaczan-tiny-vllm","name":"tiny-vllm","tagline":"Build your own high performance LLM inference engine in C++ and CUDA - a smaller version of vLLM","github_url":"https://github.com/jmaczan/tiny-vllm","owner":"jmaczan","repo":"tiny-vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/18054202?v=4","primary_language":"C++","stars":947,"forks":68,"topics":["ai","attention","batching","course","cpp","cuda","hpc","inference","llm","llm-inference","pagedattention","tiny-vllm","vllm"],"archived":false,"github_pushed_at":"2026-07-02T18:32:16+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm","markdown_url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jmaczan-tiny-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jmaczan-tiny-vllm","shared_categories":["inference-serving"]},{"slug":"harleyszhang-llm-note","name":"llm_note","tagline":"LLM notes covering model inference transformer structures and framework analysis","github_url":"https://github.com/harleyszhang/llm_note","owner":"harleyszhang","repo":"llm_note","owner_avatar_url":"https://avatars.githubusercontent.com/u/37138671?v=4","primary_language":"Python","stars":889,"forks":88,"topics":["cuda-programming","kv-cache","llm","llm-inference","transformer-models","triton-kernels","vllm"],"archived":false,"github_pushed_at":"2026-07-02T16:44:08+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/harleyszhang-llm-note","markdown_url":"https://www.graphcanon.com/tools/harleyszhang-llm-note.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/harleyszhang-llm-note","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=harleyszhang-llm-note","shared_categories":["inference-serving"]},{"slug":"ddalcu-mlx-serve","name":"mlx-serve","tagline":"Native LLM inference server for Apple Silicon","github_url":"https://github.com/ddalcu/mlx-serve","owner":"ddalcu","repo":"mlx-serve","owner_avatar_url":"https://avatars.githubusercontent.com/u/869085?v=4","primary_language":"Zig","stars":589,"forks":43,"topics":["agent","anthropic-api","apple-silicon","claude-code","deepseek-v4","diffusion","gguf","image-generation","inference","llm","local-llm","macos","macos-app","mlx","openai-api","tool-calling","video-generation","voice-agent","voice-cloning","zig"],"archived":false,"github_pushed_at":"2026-08-12T15:34:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ddalcu-mlx-serve","markdown_url":"https://www.graphcanon.com/tools/ddalcu-mlx-serve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ddalcu-mlx-serve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ddalcu-mlx-serve","shared_categories":["inference-serving"]},{"slug":"kaito-project-aikit","name":"aikit","tagline":"Fine-tune, build, and deploy open-source LLMs easily!","github_url":"https://github.com/kaito-project/aikit","owner":"kaito-project","repo":"aikit","owner_avatar_url":"https://avatars.githubusercontent.com/u/186863079?v=4","primary_language":"Go","stars":534,"forks":57,"topics":["ai","buildkit","chatgpt","docker","fine-tuning","finetuning","gemma","gpt","inference","kubernetes","large-language-models","llama","llm","localllama","mistral","mixtral","nvidia","open-llm","open-source-llm","openai"],"archived":false,"github_pushed_at":"2026-07-20T03:11:50+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/kaito-project-aikit","markdown_url":"https://www.graphcanon.com/tools/kaito-project-aikit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kaito-project-aikit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kaito-project-aikit","shared_categories":["inference-serving"]}]}}