{"data":{"node":{"slug":"quantumaikr-quant-cpp","name":"quant.cpp","tagline":"LLM inference with extended context using C","github_url":"https://github.com/quantumaikr/quant.cpp","owner":"quantumaikr","repo":"quant.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/49489033?v=4","primary_language":"C","stars":399,"forks":44,"topics":["delta-compression","embeddable","gguf","kv-cache","llm","llm-inference","pure-c","quantization","transformer","turboquant"],"archived":false,"github_pushed_at":"2026-04-26T11:15:51+00:00","maintenance_label":"Slowing","stars_delta_30d":4,"url":"https://www.graphcanon.com/tools/quantumaikr-quant-cpp","markdown_url":"https://www.graphcanon.com/tools/quantumaikr-quant-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/quantumaikr-quant-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=quantumaikr-quant-cpp"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"delta-compression","name":"delta-compression"},{"slug":"embeddable","name":"embeddable"},{"slug":"gguf","name":"gguf"},{"slug":"kv-cache","name":"kv-cache"},{"slug":"llm-inference","name":"llm-inference"},{"slug":"pure-c","name":"pure-c"},{"slug":"quantization","name":"quantization"},{"slug":"transformer","name":"transformer"}],"edges":[],"neighbours":[{"slug":"ggml-org-llama-cpp","name":"llama.cpp","tagline":"LLM inference in C/C++","github_url":"https://github.com/ggml-org/llama.cpp","owner":"ggml-org","repo":"llama.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/134263123?v=4","primary_language":"C++","stars":122941,"forks":21406,"topics":["ggml"],"archived":false,"github_pushed_at":"2026-08-07T05:28:54+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp","markdown_url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ggml-org-llama-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ggml-org-llama-cpp","shared_categories":["inference-serving"]},{"slug":"lyogavin-airllm","name":"airllm","tagline":"AirLLM 70B inference with single 4GB GPU","github_url":"https://github.com/lyogavin/airllm","owner":"lyogavin","repo":"airllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/1113905?v=4","primary_language":"Jupyter Notebook","stars":24183,"forks":2722,"topics":["chinese-llm","chinese-nlp","finetune","generative-ai","instruct-gpt","instruction-set","llama","llm","lora","open-models","open-source","open-source-models","qlora"],"archived":false,"github_pushed_at":"2026-07-23T08:29:43+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lyogavin-airllm","markdown_url":"https://www.graphcanon.com/tools/lyogavin-airllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lyogavin-airllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lyogavin-airllm","shared_categories":["inference-serving"]},{"slug":"artidoro-qlora","name":"qlora","tagline":"QLoRA finetuning of quantized LLMs","github_url":"https://github.com/artidoro/qlora","owner":"artidoro","repo":"qlora","owner_avatar_url":"https://avatars.githubusercontent.com/u/11949572?v=4","primary_language":"Jupyter Notebook","stars":10979,"forks":876,"topics":[],"archived":false,"github_pushed_at":"2024-06-10T19:20:16+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/artidoro-qlora","markdown_url":"https://www.graphcanon.com/tools/artidoro-qlora.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/artidoro-qlora","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=artidoro-qlora","shared_categories":[]},{"slug":"bitsandbytes-foundation-bitsandbytes","name":"bitsandbytes","tagline":"Large language model quantization toolkit for PyTorch.","github_url":"https://github.com/bitsandbytes-foundation/bitsandbytes","owner":"bitsandbytes-foundation","repo":"bitsandbytes","owner_avatar_url":"https://avatars.githubusercontent.com/u/175231607?v=4","primary_language":"Python","stars":8385,"forks":900,"topics":["llm","machine-learning","pytorch","qlora","quantization"],"archived":false,"github_pushed_at":"2026-07-29T18:27:51+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes","markdown_url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bitsandbytes-foundation-bitsandbytes","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bitsandbytes-foundation-bitsandbytes","shared_categories":["inference-serving"]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["inference-serving"]},{"slug":"xlite-dev-awesome-llm-inference","name":"Awesome-LLM-Inference","tagline":"A curated list of LLM/VLM inference papers with codes","github_url":"https://github.com/xlite-dev/Awesome-LLM-Inference","owner":"xlite-dev","repo":"Awesome-LLM-Inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/204302598?v=4","primary_language":"Python","stars":5477,"forks":429,"topics":["awesome-llm","deepseek","deepseek-r1","deepseek-v3","flash-attention","flash-attention-3","flash-mla","llm-inference","minimax-01","mla","paged-attention","qwen3","tensorrt-llm","vllm"],"archived":false,"github_pushed_at":"2026-08-14T12:23:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference","markdown_url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xlite-dev-awesome-llm-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xlite-dev-awesome-llm-inference","shared_categories":["inference-serving"]},{"slug":"yvgude-lean-ctx","name":"lean-ctx","tagline":"Control what your AI can see by serving context with a local Rust binary.","github_url":"https://github.com/yvgude/lean-ctx","owner":"yvgude","repo":"lean-ctx","owner_avatar_url":"https://avatars.githubusercontent.com/u/7590809?v=4","primary_language":"Rust","stars":3486,"forks":312,"topics":["agentic-coding","ai","ai-agents","ai-coding","claude-code","context-engineering","context-intelligence","context-layer","copilot","cursor","developer-tools","gemini-cli","lean-context","llm","mcp","mcp-server","reduce-token-costs","rust","token-optimization"],"archived":false,"github_pushed_at":"2026-08-04T11:40:56+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/yvgude-lean-ctx","markdown_url":"https://www.graphcanon.com/tools/yvgude-lean-ctx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/yvgude-lean-ctx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=yvgude-lean-ctx","shared_categories":[]},{"slug":"turboderp-exllama","name":"exllama","tagline":"Memory-efficient rewrite of HF transformers for Llama with quantized weights","github_url":"https://github.com/turboderp/exllama","owner":"turboderp","repo":"exllama","owner_avatar_url":"https://avatars.githubusercontent.com/u/11859846?v=4","primary_language":"Python","stars":2937,"forks":220,"topics":[],"archived":false,"github_pushed_at":"2023-09-30T19:06:04+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/turboderp-exllama","markdown_url":"https://www.graphcanon.com/tools/turboderp-exllama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/turboderp-exllama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=turboderp-exllama","shared_categories":["inference-serving"]},{"slug":"ngxson-wllama","name":"wllama","tagline":"WebAssembly binding for llama.cpp - Enabling on-browser LLM inference","github_url":"https://github.com/ngxson/wllama","owner":"ngxson","repo":"wllama","owner_avatar_url":"https://avatars.githubusercontent.com/u/7702203?v=4","primary_language":"TypeScript","stars":1159,"forks":117,"topics":["llama","llamacpp","llm","wasm","webassembly"],"archived":false,"github_pushed_at":"2026-06-17T17:32:59+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/ngxson-wllama","markdown_url":"https://www.graphcanon.com/tools/ngxson-wllama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ngxson-wllama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ngxson-wllama","shared_categories":["inference-serving"]},{"slug":"jmaczan-tiny-vllm","name":"tiny-vllm","tagline":"Build your own high performance LLM inference engine in C++ and CUDA - a smaller version of vLLM","github_url":"https://github.com/jmaczan/tiny-vllm","owner":"jmaczan","repo":"tiny-vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/18054202?v=4","primary_language":"C++","stars":1075,"forks":84,"topics":["ai","attention","batching","course","cpp","cuda","hpc","inference","llm","llm-inference","pagedattention","tiny-vllm","vllm"],"archived":false,"github_pushed_at":"2026-08-23T14:20:13+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm","markdown_url":"https://www.graphcanon.com/tools/jmaczan-tiny-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jmaczan-tiny-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jmaczan-tiny-vllm","shared_categories":["inference-serving"]},{"slug":"foldl-chatllm-cpp","name":"chatllm.cpp","tagline":"C++ real-time chat models for CPU and GPU","github_url":"https://github.com/foldl/chatllm.cpp","owner":"foldl","repo":"chatllm.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/4046440?v=4","primary_language":"C++","stars":917,"forks":72,"topics":["llm","llm-inference"],"archived":false,"github_pushed_at":"2026-08-22T08:30:42+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/foldl-chatllm-cpp","markdown_url":"https://www.graphcanon.com/tools/foldl-chatllm-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/foldl-chatllm-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=foldl-chatllm-cpp","shared_categories":["inference-serving"]},{"slug":"harleyszhang-llm-note","name":"llm_note","tagline":"LLM notes covering model inference transformer structures and framework analysis","github_url":"https://github.com/harleyszhang/llm_note","owner":"harleyszhang","repo":"llm_note","owner_avatar_url":"https://avatars.githubusercontent.com/u/37138671?v=4","primary_language":"Python","stars":888,"forks":90,"topics":["cuda-programming","kv-cache","llm","llm-inference","transformer-models","triton-kernels","vllm"],"archived":false,"github_pushed_at":"2026-08-19T06:46:41+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/harleyszhang-llm-note","markdown_url":"https://www.graphcanon.com/tools/harleyszhang-llm-note.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/harleyszhang-llm-note","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=harleyszhang-llm-note","shared_categories":["inference-serving"]}]}}