{"data":{"node":{"slug":"foldl-chatllm-cpp","name":"chatllm.cpp","tagline":"C++ real-time chat models for CPU and GPU","github_url":"https://github.com/foldl/chatllm.cpp","owner":"foldl","repo":"chatllm.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/4046440?v=4","primary_language":"C++","stars":917,"forks":72,"topics":["llm","llm-inference"],"archived":false,"github_pushed_at":"2026-08-22T08:30:42+00:00","maintenance_label":"Very active","stars_delta_30d":5,"url":"https://www.graphcanon.com/tools/foldl-chatllm-cpp","markdown_url":"https://www.graphcanon.com/tools/foldl-chatllm-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/foldl-chatllm-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=foldl-chatllm-cpp"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"},{"slug":"llm-frameworks","name":"LLM Frameworks","url":"https://www.graphcanon.com/categories/llm-frameworks","markdown_url":"https://www.graphcanon.com/categories/llm-frameworks.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/llm-frameworks"}],"tags":[{"slug":"cpu-support","name":"cpu-support"},{"slug":"gpu-support","name":"gpu-support"},{"slug":"llm","name":"llm"},{"slug":"llm-inference","name":"llm-inference"},{"slug":"real-time-chatting","name":"real-time-chatting"}],"edges":[],"neighbours":[{"slug":"ggml-org-llama-cpp","name":"llama.cpp","tagline":"LLM inference in C/C++","github_url":"https://github.com/ggml-org/llama.cpp","owner":"ggml-org","repo":"llama.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/134263123?v=4","primary_language":"C++","stars":122941,"forks":21406,"topics":["ggml"],"archived":false,"github_pushed_at":"2026-08-07T05:28:54+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp","markdown_url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ggml-org-llama-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ggml-org-llama-cpp","shared_categories":["inference-serving"]},{"slug":"ggml-org-whisper-cpp","name":"whisper.cpp","tagline":"Port of OpenAI's Whisper model in C/C++ for speech-to-text inference","github_url":"https://github.com/ggml-org/whisper.cpp","owner":"ggml-org","repo":"whisper.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/134263123?v=4","primary_language":"C++","stars":52501,"forks":5971,"topics":["inference","openai","speech-recognition","speech-to-text","transformer","whisper"],"archived":false,"github_pushed_at":"2026-07-31T07:11:28+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ggml-org-whisper-cpp","markdown_url":"https://www.graphcanon.com/tools/ggml-org-whisper-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ggml-org-whisper-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ggml-org-whisper-cpp","shared_categories":["inference-serving"]},{"slug":"gaizhenbiao-chuanhuchatgpt","name":"ChuanhuChatGPT","tagline":"GUI for ChatGPT and LLMs with features like agents and file-based QA.","github_url":"https://github.com/GaiZhenbiao/ChuanhuChatGPT","owner":"GaiZhenbiao","repo":"ChuanhuChatGPT","owner_avatar_url":"https://avatars.githubusercontent.com/u/51039745?v=4","primary_language":"Python","stars":15288,"forks":2209,"topics":["chatbot","chatglm","chatgpt-api","claude","dalle3","ernie","gemini","gemma","inspurai","llama","midjourney","minimax","moss","ollama","qwen","spark","stablelm"],"archived":false,"github_pushed_at":"2026-04-30T13:54:58+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/gaizhenbiao-chuanhuchatgpt","markdown_url":"https://www.graphcanon.com/tools/gaizhenbiao-chuanhuchatgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/gaizhenbiao-chuanhuchatgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=gaizhenbiao-chuanhuchatgpt","shared_categories":["inference-serving"]},{"slug":"lightning-ai-litgpt","name":"litgpt","tagline":"High-performance LLMs with recipes for pretraining, finetuning and deployment","github_url":"https://github.com/Lightning-AI/litgpt","owner":"Lightning-AI","repo":"litgpt","owner_avatar_url":"https://avatars.githubusercontent.com/u/58386951?v=4","primary_language":"Python","stars":13605,"forks":1483,"topics":["ai","artificial-intelligence","deep-learning","large-language-models","llm","llm-inference","llms"],"archived":false,"github_pushed_at":"2026-07-20T10:24:12+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lightning-ai-litgpt","markdown_url":"https://www.graphcanon.com/tools/lightning-ai-litgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lightning-ai-litgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lightning-ai-litgpt","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"simonw-llm","name":"llm","tagline":"Access large language models from the command-line","github_url":"https://github.com/simonw/llm","owner":"simonw","repo":"llm","owner_avatar_url":"https://avatars.githubusercontent.com/u/9599?v=4","primary_language":"Python","stars":12324,"forks":939,"topics":["ai","llms","openai"],"archived":false,"github_pushed_at":"2026-08-05T14:29:09+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/simonw-llm","markdown_url":"https://www.graphcanon.com/tools/simonw-llm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/simonw-llm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=simonw-llm","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"oumi-ai-oumi","name":"oumi","tagline":"Easily fine-tune, evaluate and deploy open source LLMs/VLMs","github_url":"https://github.com/oumi-ai/oumi","owner":"oumi-ai","repo":"oumi","owner_avatar_url":"https://avatars.githubusercontent.com/u/167452922?v=4","primary_language":"Python","stars":9376,"forks":784,"topics":["dpo","evaluation","fine-tuning","gpt-oss","gpt-oss-120b","gpt-oss-20b","inference","llama","llms","open-weight","open-weight-models","open-weights","sft","slms","vlms"],"archived":false,"github_pushed_at":"2026-08-21T23:11:35+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/oumi-ai-oumi","markdown_url":"https://www.graphcanon.com/tools/oumi-ai-oumi.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/oumi-ai-oumi","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=oumi-ai-oumi","shared_categories":["inference-serving"]},{"slug":"fareedkhan-dev-train-llm-from-scratch","name":"train-llm-from-scratch","tagline":"A straightforward method for training your LLM from raw text to aligned model generation","github_url":"https://github.com/FareedKhan-dev/train-llm-from-scratch","owner":"FareedKhan-dev","repo":"train-llm-from-scratch","owner_avatar_url":"https://avatars.githubusercontent.com/u/63067900?v=4","primary_language":"Python","stars":9141,"forks":1264,"topics":["gemini","large-language-models","llm","openai","training","transformers"],"archived":false,"github_pushed_at":"2026-08-17T05:07:26+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/fareedkhan-dev-train-llm-from-scratch","markdown_url":"https://www.graphcanon.com/tools/fareedkhan-dev-train-llm-from-scratch.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fareedkhan-dev-train-llm-from-scratch","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fareedkhan-dev-train-llm-from-scratch","shared_categories":["inference-serving"]},{"slug":"liltom-eth-llama2-webui","name":"llama2-webui","tagline":"Run Llama 2 locally with gradio UI on GPU or CPU","github_url":"https://github.com/liltom-eth/llama2-webui","owner":"liltom-eth","repo":"llama2-webui","owner_avatar_url":"https://avatars.githubusercontent.com/u/11456256?v=4","primary_language":"Jupyter Notebook","stars":1937,"forks":199,"topics":["llama-2","llama2","llm","llm-inference"],"archived":false,"github_pushed_at":"2024-03-22T09:50:24+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/liltom-eth-llama2-webui","markdown_url":"https://www.graphcanon.com/tools/liltom-eth-llama2-webui.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/liltom-eth-llama2-webui","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=liltom-eth-llama2-webui","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"sauravpanda-browserai","name":"BrowserAI","tagline":"Run local LLMs like llama, deepseek-distill, kokoro and more inside your browser","github_url":"https://github.com/sauravpanda/BrowserAI","owner":"sauravpanda","repo":"BrowserAI","owner_avatar_url":"https://avatars.githubusercontent.com/u/12201824?v=4","primary_language":"TypeScript","stars":1449,"forks":138,"topics":["agents","ai","llama","llm","llm-inference","local","localllm","tts","webgpu"],"archived":false,"github_pushed_at":"2026-07-21T02:47:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/sauravpanda-browserai","markdown_url":"https://www.graphcanon.com/tools/sauravpanda-browserai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/sauravpanda-browserai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=sauravpanda-browserai","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"jmaczan-tiny-vllm","name":"tiny-vllm","tagline":"Build your own high performance LLM inference engine in C++ and CUDA - a smaller version of vLLM","github_url":"https://github.com/jmaczan/tiny-vllm","owner":"jmaczan","repo":"tiny-vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/18054202?v=4","primary_language":"C++","stars":1075,"forks":84,"topics":["ai","attention","batching","course","cpp","cuda","hpc","inference","llm","llm-inference","pagedattention","tiny-vllm","vllm"],"archived":false,"github_pushed_at":"2026-08-23T14:20:13+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm","markdown_url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jmaczan-tiny-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jmaczan-tiny-vllm","shared_categories":["inference-serving"]},{"slug":"keyvank-femtogpt","name":"femtoGPT","tagline":"Pure Rust implementation of a minimal Generative Pretrained Transformer","github_url":"https://github.com/keyvank/femtoGPT","owner":"keyvank","repo":"femtoGPT","owner_avatar_url":"https://avatars.githubusercontent.com/u/4275654?v=4","primary_language":"Rust","stars":935,"forks":67,"topics":["from-scratch","gpt","gpu","hacktoberfest","llm","machine-learning","neural-network","opencl","rust"],"archived":false,"github_pushed_at":"2025-10-21T11:13:42+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/keyvank-femtogpt","markdown_url":"https://www.graphcanon.com/tools/keyvank-femtogpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/keyvank-femtogpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=keyvank-femtogpt","shared_categories":["llm-frameworks"]},{"slug":"openinfer-project-openinfer","name":"openinfer","tagline":"Pure Rust CUDA LLM inference engine serving multiple models including Qwen3 and Kimi-K2","github_url":"https://github.com/openinfer-project/openinfer","owner":"openinfer-project","repo":"openinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/292134277?v=4","primary_language":"Rust","stars":657,"forks":103,"topics":["cuda","cuda-kernels","deepseek","gpu","inference","inference-engine","kimi","kimi-k2","kv-cache","llm","llm-inference","llm-serving","model-serving","moe","openai-api","paged-attention","qwen","qwen3","rust","vllm"],"archived":false,"github_pushed_at":"2026-08-25T05:44:30+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/openinfer-project-openinfer","markdown_url":"https://www.graphcanon.com/tools/openinfer-project-openinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openinfer-project-openinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openinfer-project-openinfer","shared_categories":["inference-serving"]}]}}