{"data":{"node":{"slug":"fasterdecoding-rest","name":"REST","tagline":"REST: Retrieval-Based Speculative Decoding","github_url":"https://github.com/FasterDecoding/REST","owner":"FasterDecoding","repo":"REST","owner_avatar_url":"https://avatars.githubusercontent.com/u/144572371?v=4","primary_language":"C","stars":220,"forks":17,"topics":["llm-inference","retrieval","speculative-decoding"],"archived":false,"github_pushed_at":"2026-03-05T12:38:16+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/fasterdecoding-rest","markdown_url":"https://www.graphcanon.com/tools/fasterdecoding-rest.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fasterdecoding-rest","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fasterdecoding-rest"},"categories":[{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"},{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"llm-inference","name":"llm-inference"},{"slug":"retrieval","name":"retrieval"},{"slug":"speculative-decoding","name":"speculative-decoding"}],"edges":[],"neighbours":[{"slug":"ggml-org-llama-cpp","name":"llama.cpp","tagline":"LLM inference in C/C++","github_url":"https://github.com/ggml-org/llama.cpp","owner":"ggml-org","repo":"llama.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/134263123?v=4","primary_language":"C++","stars":122941,"forks":21406,"topics":["ggml"],"archived":false,"github_pushed_at":"2026-08-07T05:28:54+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp","markdown_url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ggml-org-llama-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ggml-org-llama-cpp","shared_categories":["inference-serving"]},{"slug":"nirdiamant-rag-techniques","name":"RAG_Techniques","tagline":"Showcases advanced techniques for Retrieval-Augmented Generation (RAG) systems with detailed notebook tutorials.","github_url":"https://github.com/NirDiamant/RAG_Techniques","owner":"NirDiamant","repo":"RAG_Techniques","owner_avatar_url":"https://avatars.githubusercontent.com/u/28316913?v=4","primary_language":"Jupyter Notebook","stars":29076,"forks":3540,"topics":["agentic-rag","ai","embeddings","generative-ai","gpt","langchain","llama-index","llm","llms","machine-learning","nlp","openai","python","rag","retrieval-augmented-generation","semantic-search","tutorials","vector-database"],"archived":false,"github_pushed_at":"2026-08-15T00:52:05+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/nirdiamant-rag-techniques","markdown_url":"https://www.graphcanon.com/tools/nirdiamant-rag-techniques.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nirdiamant-rag-techniques","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nirdiamant-rag-techniques","shared_categories":["data-retrieval"]},{"slug":"lyogavin-airllm","name":"airllm","tagline":"AirLLM 70B inference with single 4GB GPU","github_url":"https://github.com/lyogavin/airllm","owner":"lyogavin","repo":"airllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/1113905?v=4","primary_language":"Jupyter Notebook","stars":24183,"forks":2722,"topics":["chinese-llm","chinese-nlp","finetune","generative-ai","instruct-gpt","instruction-set","llama","llm","lora","open-models","open-source","open-source-models","qlora"],"archived":false,"github_pushed_at":"2026-07-23T08:29:43+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lyogavin-airllm","markdown_url":"https://www.graphcanon.com/tools/lyogavin-airllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lyogavin-airllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lyogavin-airllm","shared_categories":["inference-serving"]},{"slug":"ericlbuehler-mistral-rs","name":"mistral.rs","tagline":"Fast flexible LLM inference","github_url":"https://github.com/EricLBuehler/mistral.rs","owner":"EricLBuehler","repo":"mistral.rs","owner_avatar_url":"https://avatars.githubusercontent.com/u/65165915?v=4","primary_language":"Rust","stars":7575,"forks":671,"topics":["llm","rust","uqff"],"archived":false,"github_pushed_at":"2026-07-29T20:21:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs","markdown_url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ericlbuehler-mistral-rs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ericlbuehler-mistral-rs","shared_categories":["inference-serving"]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["inference-serving"]},{"slug":"xlite-dev-awesome-llm-inference","name":"Awesome-LLM-Inference","tagline":"A curated list of LLM/VLM inference papers with codes","github_url":"https://github.com/xlite-dev/Awesome-LLM-Inference","owner":"xlite-dev","repo":"Awesome-LLM-Inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/204302598?v=4","primary_language":"Python","stars":5477,"forks":429,"topics":["awesome-llm","deepseek","deepseek-r1","deepseek-v3","flash-attention","flash-attention-3","flash-mla","llm-inference","minimax-01","mla","paged-attention","qwen3","tensorrt-llm","vllm"],"archived":false,"github_pushed_at":"2026-08-14T12:23:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference","markdown_url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xlite-dev-awesome-llm-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xlite-dev-awesome-llm-inference","shared_categories":["inference-serving"]},{"slug":"marker-inc-korea-autorag","name":"AutoRAG","tagline":"Open-source framework for RAG evaluation and optimization via AutoML","github_url":"https://github.com/Marker-Inc-Korea/AutoRAG","owner":"Marker-Inc-Korea","repo":"AutoRAG","owner_avatar_url":"https://avatars.githubusercontent.com/u/74290595?v=4","primary_language":"TypeScript","stars":4968,"forks":419,"topics":["analysis","automl","benchmarking","document-parser","embeddings","evaluation","llm","llm-evaluation","llm-ops","open-source","ops","optimization","pipeline","python","qa","rag","rag-evaluation","retrieval-augmented-generation"],"archived":false,"github_pushed_at":"2026-08-05T14:46:13+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/marker-inc-korea-autorag","markdown_url":"https://www.graphcanon.com/tools/marker-inc-korea-autorag.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/marker-inc-korea-autorag","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=marker-inc-korea-autorag","shared_categories":[]},{"slug":"fasterdecoding-medusa","name":"Medusa","tagline":"Framework for accelerating LLM generation using multiple decoding heads","github_url":"https://github.com/FasterDecoding/Medusa","owner":"FasterDecoding","repo":"Medusa","owner_avatar_url":"https://avatars.githubusercontent.com/u/144572371?v=4","primary_language":"Jupyter Notebook","stars":2767,"forks":205,"topics":["llm","llm-inference"],"archived":false,"github_pushed_at":"2024-06-25T12:23:04+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/fasterdecoding-medusa","markdown_url":"https://www.graphcanon.com/tools/fasterdecoding-medusa.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fasterdecoding-medusa","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fasterdecoding-medusa","shared_categories":["inference-serving"]},{"slug":"huangowen-awesome-llm-compression","name":"Awesome-LLM-Compression","tagline":"Awesome LLM compression research papers and tools to accelerate LLM training and inference.","github_url":"https://github.com/HuangOwen/Awesome-LLM-Compression","owner":"HuangOwen","repo":"Awesome-LLM-Compression","owner_avatar_url":"https://avatars.githubusercontent.com/u/24937399?v=4","primary_language":null,"stars":1859,"forks":129,"topics":[],"archived":false,"github_pushed_at":"2026-06-30T15:26:46+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression","markdown_url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huangowen-awesome-llm-compression","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huangowen-awesome-llm-compression","shared_categories":["inference-serving"]},{"slug":"parthsarthi03-raptor","name":"raptor","tagline":"Recursive Abstractive Processing for Tree-Organized Retrieval","github_url":"https://github.com/parthsarthi03/raptor","owner":"parthsarthi03","repo":"raptor","owner_avatar_url":"https://avatars.githubusercontent.com/u/39787228?v=4","primary_language":"Python","stars":1742,"forks":233,"topics":["agents","clustering","framework","language-model","llm","machine-learning","rag","retrieval","retrieval-augmented-generation","vector-database"],"archived":false,"github_pushed_at":"2024-09-03T08:34:31+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/parthsarthi03-raptor","markdown_url":"https://www.graphcanon.com/tools/parthsarthi03-raptor.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/parthsarthi03-raptor","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=parthsarthi03-raptor","shared_categories":[]},{"slug":"jzbjyb-flare","name":"FLARE","tagline":"Forward-Looking Active REtrieval-augmented generation","github_url":"https://github.com/jzbjyb/FLARE","owner":"jzbjyb","repo":"FLARE","owner_avatar_url":"https://avatars.githubusercontent.com/u/5134761?v=4","primary_language":"Python","stars":670,"forks":62,"topics":[],"archived":false,"github_pushed_at":"2023-11-20T08:25:17+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/jzbjyb-flare","markdown_url":"https://www.graphcanon.com/tools/jzbjyb-flare.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jzbjyb-flare","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jzbjyb-flare","shared_categories":["data-retrieval"]},{"slug":"denis2054-rag-driven-generative-ai","name":"RAG-Driven-Generative-AI","tagline":"Builds Retrieval Augmented Generation AI using LlamaIndex with support from Deep Lake and Pinecone","github_url":"https://github.com/Denis2054/RAG-Driven-Generative-AI","owner":"Denis2054","repo":"RAG-Driven-Generative-AI","owner_avatar_url":"https://avatars.githubusercontent.com/u/30811222?v=4","primary_language":"Jupyter Notebook","stars":621,"forks":215,"topics":["advanced-rag","chroma","chromadb","embedding-models","fine-tuning","gpt-4o-mini","gpt4-omni","grok","huggingface","indexing-querying","llama","llama-index","multimodal","openai-api","pinecone","rag","scaling","vision-transformer","xai-grok"],"archived":false,"github_pushed_at":"2025-09-23T15:31:25+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/denis2054-rag-driven-generative-ai","markdown_url":"https://www.graphcanon.com/tools/denis2054-rag-driven-generative-ai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/denis2054-rag-driven-generative-ai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=denis2054-rag-driven-generative-ai","shared_categories":["data-retrieval"]}]}}