{"data":{"node":{"slug":"huawei-csl-kvarn","name":"KVarN","tagline":"vLLM KV-cache quantization backend for AI agents","github_url":"https://github.com/huawei-csl/KVarN","owner":"huawei-csl","repo":"KVarN","owner_avatar_url":"https://avatars.githubusercontent.com/u/234563461?v=4","primary_language":"Python","stars":470,"forks":35,"topics":["agentic-ai","kv-cache","llm","llm-inference","long-context","quantization","vllm"],"archived":false,"github_pushed_at":"2026-06-22T09:18:28+00:00","maintenance_label":"Steady","stars_delta_30d":28,"url":"https://www.graphcanon.com/tools/huawei-csl-kvarn","markdown_url":"https://www.graphcanon.com/tools/huawei-csl-kvarn.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huawei-csl-kvarn","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huawei-csl-kvarn"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"agentic-ai","name":"agentic-ai"},{"slug":"kv-cache","name":"kv-cache"},{"slug":"llm-inference","name":"llm-inference"},{"slug":"long-context","name":"long-context"},{"slug":"quantization","name":"quantization"},{"slug":"vllm","name":"vllm"}],"edges":[],"neighbours":[{"slug":"vllm-project-vllm","name":"vllm","tagline":"A high-throughput and memory-efficient inference and serving engine for LLMs","github_url":"https://github.com/vllm-project/vllm","owner":"vllm-project","repo":"vllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/136984999?v=4","primary_language":"Python","stars":87847,"forks":20135,"topics":["amd","blackwell","cuda","deepseek","deepseek-v3","gpt","gpt-oss","inference","kimi","llama","llm","llm-serving","model-serving","moe","openai","pytorch","qwen","qwen3","tpu","transformer"],"archived":false,"github_pushed_at":"2026-08-01T11:55:36+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/vllm-project-vllm","markdown_url":"https://www.graphcanon.com/tools/vllm-project-vllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vllm-project-vllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vllm-project-vllm","shared_categories":["inference-serving"]},{"slug":"headroomlabs-ai-headroom","name":"headroom","tagline":"Compress tool outputs and data to reduce tokens before reaching the LLM.","github_url":"https://github.com/headroomlabs-ai/headroom","owner":"headroomlabs-ai","repo":"headroom","owner_avatar_url":"https://avatars.githubusercontent.com/u/294291659?v=4","primary_language":"Python","stars":66470,"forks":5103,"topics":["agent","ai","anthropic","claude-code","compression","context-engineering","context-window","cursor","fastapi","langchain","llm","mcp","openai","prompt-engineering","proxy","python","rag","token-optimization","tokens","typescript"],"archived":false,"github_pushed_at":"2026-08-16T00:57:42+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/headroomlabs-ai-headroom","markdown_url":"https://www.graphcanon.com/tools/headroomlabs-ai-headroom.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/headroomlabs-ai-headroom","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=headroomlabs-ai-headroom","shared_categories":[]},{"slug":"microsoft-semantic-kernel","name":"semantic-kernel","tagline":"Integrate cutting-edge LLM technology quickly and easily into your apps","github_url":"https://github.com/microsoft/semantic-kernel","owner":"microsoft","repo":"semantic-kernel","owner_avatar_url":"https://avatars.githubusercontent.com/u/6154722?v=4","primary_language":"C#","stars":28427,"forks":4707,"topics":["ai","artificial-intelligence","llm","openai","sdk"],"archived":false,"github_pushed_at":"2026-08-06T09:54:50+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/microsoft-semantic-kernel","markdown_url":"https://www.graphcanon.com/tools/microsoft-semantic-kernel.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/microsoft-semantic-kernel","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=microsoft-semantic-kernel","shared_categories":[]},{"slug":"lyogavin-airllm","name":"airllm","tagline":"AirLLM 70B inference with single 4GB GPU","github_url":"https://github.com/lyogavin/airllm","owner":"lyogavin","repo":"airllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/1113905?v=4","primary_language":"Jupyter Notebook","stars":24183,"forks":2722,"topics":["chinese-llm","chinese-nlp","finetune","generative-ai","instruct-gpt","instruction-set","llama","llm","lora","open-models","open-source","open-source-models","qlora"],"archived":false,"github_pushed_at":"2026-07-23T08:29:43+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lyogavin-airllm","markdown_url":"https://www.graphcanon.com/tools/lyogavin-airllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lyogavin-airllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lyogavin-airllm","shared_categories":["inference-serving"]},{"slug":"artidoro-qlora","name":"qlora","tagline":"QLoRA finetuning of quantized LLMs","github_url":"https://github.com/artidoro/qlora","owner":"artidoro","repo":"qlora","owner_avatar_url":"https://avatars.githubusercontent.com/u/11949572?v=4","primary_language":"Jupyter Notebook","stars":10979,"forks":876,"topics":[],"archived":false,"github_pushed_at":"2024-06-10T19:20:16+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/artidoro-qlora","markdown_url":"https://www.graphcanon.com/tools/artidoro-qlora.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/artidoro-qlora","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=artidoro-qlora","shared_categories":[]},{"slug":"bitsandbytes-foundation-bitsandbytes","name":"bitsandbytes","tagline":"Large language model quantization toolkit for PyTorch.","github_url":"https://github.com/bitsandbytes-foundation/bitsandbytes","owner":"bitsandbytes-foundation","repo":"bitsandbytes","owner_avatar_url":"https://avatars.githubusercontent.com/u/175231607?v=4","primary_language":"Python","stars":8385,"forks":900,"topics":["llm","machine-learning","pytorch","qlora","quantization"],"archived":false,"github_pushed_at":"2026-07-29T18:27:51+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes","markdown_url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bitsandbytes-foundation-bitsandbytes","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bitsandbytes-foundation-bitsandbytes","shared_categories":["inference-serving"]},{"slug":"zilliztech-gptcache","name":"GPTCache","tagline":"Semantic cache for LLMs.","github_url":"https://github.com/zilliztech/GPTCache","owner":"zilliztech","repo":"GPTCache","owner_avatar_url":"https://avatars.githubusercontent.com/u/18416694?v=4","primary_language":"Python","stars":8124,"forks":589,"topics":["aigc","autogpt","babyagi","chatbot","chatgpt","chatgpt-api","dolly","gpt","langchain","llama","llama-index","llm","memcache","milvus","openai","redis","semantic-search","similarity-search","vector-search"],"archived":false,"github_pushed_at":"2025-07-11T09:04:36+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/zilliztech-gptcache","markdown_url":"https://www.graphcanon.com/tools/zilliztech-gptcache.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zilliztech-gptcache","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zilliztech-gptcache","shared_categories":["inference-serving"]},{"slug":"linkedin-liger-kernel","name":"Liger-Kernel","tagline":"Efficient Triton Kernels for LLM Training","github_url":"https://github.com/linkedin/Liger-Kernel","owner":"linkedin","repo":"Liger-Kernel","owner_avatar_url":"https://avatars.githubusercontent.com/u/357098?v=4","primary_language":"Python","stars":6555,"forks":573,"topics":["finetuning","gemma2","hacktoberfest","llama","llama3","llm-training","llms","mistral","phi3","triton","triton-kernels"],"archived":false,"github_pushed_at":"2026-08-07T08:48:09+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/linkedin-liger-kernel","markdown_url":"https://www.graphcanon.com/tools/linkedin-liger-kernel.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/linkedin-liger-kernel","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=linkedin-liger-kernel","shared_categories":[]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["inference-serving"]},{"slug":"yvgude-lean-ctx","name":"lean-ctx","tagline":"Control what your AI can see by serving context with a local Rust binary.","github_url":"https://github.com/yvgude/lean-ctx","owner":"yvgude","repo":"lean-ctx","owner_avatar_url":"https://avatars.githubusercontent.com/u/7590809?v=4","primary_language":"Rust","stars":3486,"forks":312,"topics":["agentic-coding","ai","ai-agents","ai-coding","claude-code","context-engineering","context-intelligence","context-layer","copilot","cursor","developer-tools","gemini-cli","lean-context","llm","mcp","mcp-server","reduce-token-costs","rust","token-optimization"],"archived":false,"github_pushed_at":"2026-08-04T11:40:56+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/yvgude-lean-ctx","markdown_url":"https://www.graphcanon.com/tools/yvgude-lean-ctx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/yvgude-lean-ctx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=yvgude-lean-ctx","shared_categories":[]},{"slug":"turboderp-exllama","name":"exllama","tagline":"Memory-efficient rewrite of HF transformers for Llama with quantized weights","github_url":"https://github.com/turboderp/exllama","owner":"turboderp","repo":"exllama","owner_avatar_url":"https://avatars.githubusercontent.com/u/11859846?v=4","primary_language":"Python","stars":2937,"forks":220,"topics":[],"archived":false,"github_pushed_at":"2023-09-30T19:06:04+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/turboderp-exllama","markdown_url":"https://www.graphcanon.com/tools/turboderp-exllama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/turboderp-exllama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=turboderp-exllama","shared_categories":["inference-serving"]},{"slug":"kubeflow-trainer","name":"trainer","tagline":"Distributed AI Model Training and LLM Fine-Tuning on Kubernetes","github_url":"https://github.com/kubeflow/trainer","owner":"kubeflow","repo":"trainer","owner_avatar_url":"https://avatars.githubusercontent.com/u/33164907?v=4","primary_language":"Go","stars":2196,"forks":1030,"topics":["ai","distributed","fine-tuning","gpu","huggingface","jax","kubeflow","kubernetes","llm","machine-learning","mlops","python","pytorch","tensorflow","xgboost"],"archived":false,"github_pushed_at":"2026-08-22T02:27:28+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/kubeflow-trainer","markdown_url":"https://www.graphcanon.com/tools/kubeflow-trainer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kubeflow-trainer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kubeflow-trainer","shared_categories":[]}]}}