{"data":{"node":{"slug":"nightmean-ollitert","name":"OlliteRT","tagline":"Android-based local inference server for OpenAI-compatible LLMs","github_url":"https://github.com/NightMean/OlliteRT","owner":"NightMean","repo":"OlliteRT","owner_avatar_url":"https://avatars.githubusercontent.com/u/5726996?v=4","primary_language":"Kotlin","stars":373,"forks":47,"topics":["android","anthropic-api","gemma","home-assistant","kotlin-android","litert","litert-lm","llm","llm-inference","local-llm","on-device-ai","openai-api"],"archived":false,"github_pushed_at":"2026-09-19T23:57:14+00:00","maintenance_label":"Very active","stars_delta_30d":227,"url":"https://www.graphcanon.com/tools/nightmean-ollitert","markdown_url":"https://www.graphcanon.com/tools/nightmean-ollitert.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nightmean-ollitert","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nightmean-ollitert"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"android","name":"android"},{"slug":"kotlin-android","name":"kotlin-android"},{"slug":"litert-lm","name":"litert-lm"},{"slug":"local-llm","name":"local-llm"},{"slug":"on-device-ai","name":"on-device-ai"},{"slug":"openai-api","name":"openai-api"}],"edges":[],"neighbours":[{"slug":"ollama-ollama","name":"ollama","tagline":"Get up and running with Kimi, GLM, MiniMax, DeepSeek, gpt-oss, Qwen, Gemma and other models.","github_url":"https://github.com/ollama/ollama","owner":"ollama","repo":"ollama","owner_avatar_url":"https://avatars.githubusercontent.com/u/151674099?v=4","primary_language":"Go","stars":181196,"forks":17922,"topics":["deepseek","gemma","gemma3","glm","go","golang","gpt-oss","llama","llama3","llm","llms","minimax","mistral","ollama","qwen"],"archived":false,"github_pushed_at":"2026-09-17T23:54:24+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/ollama-ollama","markdown_url":"https://www.graphcanon.com/tools/ollama-ollama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ollama-ollama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ollama-ollama","shared_categories":["inference-serving"]},{"slug":"andyyyy64-whichllm","name":"whichllm","tagline":"Command-line tool to find and benchmark local LLM performance","github_url":"https://github.com/Andyyyy64/whichllm","owner":"Andyyyy64","repo":"whichllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/105579829?v=4","primary_language":"Python","stars":6666,"forks":368,"topics":["localllm"],"archived":false,"github_pushed_at":"2026-09-19T16:22:49+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/andyyyy64-whichllm","markdown_url":"https://www.graphcanon.com/tools/andyyyy64-whichllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/andyyyy64-whichllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=andyyyy64-whichllm","shared_categories":["inference-serving"]},{"slug":"osmantic-ods","name":"ODS","tagline":"Transform personal computers into AI servers.","github_url":"https://github.com/Osmantic/ODS","owner":"Osmantic","repo":"ODS","owner_avatar_url":"https://avatars.githubusercontent.com/u/262014141?v=4","primary_language":"Python","stars":6607,"forks":943,"topics":["ai-agents","amd","comfyui","docker","llama-cpp","llm","local-ai","n8n","nvidia","open-webui","rag","self-hosted","speech-to-text","strix-halo","text-to-speech","workflow-automation"],"archived":false,"github_pushed_at":"2026-09-20T07:02:35+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/osmantic-ods","markdown_url":"https://www.graphcanon.com/tools/osmantic-ods.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/osmantic-ods","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=osmantic-ods","shared_categories":["inference-serving"]},{"slug":"superlinked-sie","name":"sie","tagline":"Open-source inference server and production cluster for all the models your agent needs.","github_url":"https://github.com/superlinked/sie","owner":"superlinked","repo":"sie","owner_avatar_url":"https://avatars.githubusercontent.com/u/94243920?v=4","primary_language":"Python","stars":3293,"forks":311,"topics":["bge","colbert","data-pipeline","deep-learning","embeddings","inference","inference-server","information-retrieval","llm","ml","mlops","natural-language-processing","nlp","python","reranking","retrieval","retrieval-augmented-generation","semantic-search","splade","vector-search"],"archived":false,"github_pushed_at":"2026-09-19T01:00:02+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/superlinked-sie","markdown_url":"https://www.graphcanon.com/tools/superlinked-sie.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/superlinked-sie","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=superlinked-sie","shared_categories":["inference-serving"]},{"slug":"b4rtaz-distributed-llama","name":"distributed-llama","tagline":"Distributed LLM inference using home devices cluster","github_url":"https://github.com/b4rtaz/distributed-llama","owner":"b4rtaz","repo":"distributed-llama","owner_avatar_url":"https://avatars.githubusercontent.com/u/12797776?v=4","primary_language":"C++","stars":3060,"forks":250,"topics":["distributed-computing","distributed-llm","llama2","llama3","llm","llm-inference","llms","neural-network","open-llm"],"archived":false,"github_pushed_at":"2026-07-05T16:47:20+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama","markdown_url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/b4rtaz-distributed-llama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=b4rtaz-distributed-llama","shared_categories":["inference-serving"]},{"slug":"rafska-awesome-local-llm","name":"awesome-local-llm","tagline":"Resources for running LLMs locally","github_url":"https://github.com/rafska/awesome-local-llm","owner":"rafska","repo":"awesome-local-llm","owner_avatar_url":"https://avatars.githubusercontent.com/u/17859377?v=4","primary_language":null,"stars":2869,"forks":388,"topics":["ai","awesome","awesome-list","llm","local","local-ai","local-llm","resources","self-hosted","selfhosted"],"archived":false,"github_pushed_at":"2026-09-13T09:48:26+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/rafska-awesome-local-llm","markdown_url":"https://www.graphcanon.com/tools/rafska-awesome-local-llm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rafska-awesome-local-llm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rafska-awesome-local-llm","shared_categories":["inference-serving"]},{"slug":"jhubi1-ollama-app","name":"ollama-app","tagline":"A modern and easy-to-use client for Ollama","github_url":"https://github.com/JHubi1/ollama-app","owner":"JHubi1","repo":"ollama-app","owner_avatar_url":"https://avatars.githubusercontent.com/u/61345690?v=4","primary_language":"Dart","stars":1809,"forks":220,"topics":["ai","android","app","linux","llama","localai","ollama","ollama-client","windows"],"archived":false,"github_pushed_at":"2026-08-07T21:26:25+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/jhubi1-ollama-app","markdown_url":"https://www.graphcanon.com/tools/jhubi1-ollama-app.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jhubi1-ollama-app","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jhubi1-ollama-app","shared_categories":[]},{"slug":"waybarrios-vllm-mlx","name":"vllm-mlx","tagline":"Server for LLMs and vision-language models compatible with Apple Silicon","github_url":"https://github.com/waybarrios/vllm-mlx","owner":"waybarrios","repo":"vllm-mlx","owner_avatar_url":"https://avatars.githubusercontent.com/u/6794828?v=4","primary_language":"Python","stars":1588,"forks":222,"topics":["anthropic","anthropic-api","apple-silicon","claude-code","continuous-batching","inference-server","llm","local-llm","macos","mcp","mlx","multimodal-ai","openai","openai-api","openai-compatible","speech-to-text","text-to-speech","tool-calling","vision-language-model","vllm"],"archived":false,"github_pushed_at":"2026-09-19T18:11:07+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/waybarrios-vllm-mlx","markdown_url":"https://www.graphcanon.com/tools/waybarrios-vllm-mlx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/waybarrios-vllm-mlx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=waybarrios-vllm-mlx","shared_categories":["inference-serving"]},{"slug":"sauravpanda-browserai","name":"BrowserAI","tagline":"Run local LLMs like llama, deepseek-distill, kokoro and more inside your browser","github_url":"https://github.com/sauravpanda/BrowserAI","owner":"sauravpanda","repo":"BrowserAI","owner_avatar_url":"https://avatars.githubusercontent.com/u/12201824?v=4","primary_language":"TypeScript","stars":1451,"forks":136,"topics":["agents","ai","llama","llm","llm-inference","local","localllm","tts","webgpu"],"archived":false,"github_pushed_at":"2026-07-21T02:47:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/sauravpanda-browserai","markdown_url":"https://www.graphcanon.com/tools/sauravpanda-browserai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/sauravpanda-browserai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=sauravpanda-browserai","shared_categories":["inference-serving"]},{"slug":"ddalcu-mlx-serve","name":"mlx-serve","tagline":"Native LLM inference server for Apple Silicon","github_url":"https://github.com/ddalcu/mlx-serve","owner":"ddalcu","repo":"mlx-serve","owner_avatar_url":"https://avatars.githubusercontent.com/u/869085?v=4","primary_language":"Zig","stars":1418,"forks":130,"topics":["agent","anthropic-api","apple-silicon","claude-code","deepseek-v4","diffusion","gguf","image-generation","inference","llm","local-llm","macos","macos-app","mlx","openai-api","tool-calling","video-generation","voice-agent","voice-cloning","zig"],"archived":false,"github_pushed_at":"2026-09-19T22:08:35+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/ddalcu-mlx-serve","markdown_url":"https://www.graphcanon.com/tools/ddalcu-mlx-serve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ddalcu-mlx-serve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ddalcu-mlx-serve","shared_categories":["inference-serving"]},{"slug":"jonigl-mcp-client-for-ollama","name":"mcp-client-for-ollama","tagline":"TUI MCP Client for Ollama enables local LLM interaction with extensive features.","github_url":"https://github.com/jonigl/mcp-client-for-ollama","owner":"jonigl","repo":"mcp-client-for-ollama","owner_avatar_url":"https://avatars.githubusercontent.com/u/4612832?v=4","primary_language":"Python","stars":824,"forks":129,"topics":["agentic-ai","ai","command-line-tool","harness","linux","local-llm","macos","mcp","mcp-client","mcp-prompts","mcp-resources","mcp-server","mcp-tools","model-context-protocol","ollama","open-source","sse","stdio","streamable-http","windows"],"archived":false,"github_pushed_at":"2026-09-14T08:34:02+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jonigl-mcp-client-for-ollama","markdown_url":"https://www.graphcanon.com/tools/jonigl-mcp-client-for-ollama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jonigl-mcp-client-for-ollama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jonigl-mcp-client-for-ollama","shared_categories":["inference-serving"]},{"slug":"openinfer-project-openinfer","name":"openinfer","tagline":"Pure Rust CUDA LLM inference engine serving multiple models including Qwen3 and Kimi-K2","github_url":"https://github.com/openinfer-project/openinfer","owner":"openinfer-project","repo":"openinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/292134277?v=4","primary_language":"Rust","stars":705,"forks":107,"topics":["cuda","cuda-kernels","deepseek","gpu","inference","inference-engine","kimi","kimi-k2","kv-cache","llm","llm-inference","llm-serving","model-serving","moe","openai-api","paged-attention","qwen","qwen3","rust","vllm"],"archived":false,"github_pushed_at":"2026-09-18T17:57:56+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/openinfer-project-openinfer","markdown_url":"https://www.graphcanon.com/tools/openinfer-project-openinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openinfer-project-openinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openinfer-project-openinfer","shared_categories":["inference-serving"]}]}}