{"data":{"node":{"slug":"nvidia-star-attention","name":"Star-Attention","tagline":"Efficient LLM Inference over Long Sequences","github_url":"https://github.com/NVIDIA/Star-Attention","owner":"NVIDIA","repo":"Star-Attention","owner_avatar_url":"https://avatars.githubusercontent.com/u/1728152?v=4","primary_language":"Python","stars":392,"forks":25,"topics":["attention-mechanism","large-language-models","llm-inference"],"archived":false,"github_pushed_at":"2025-06-25T19:36:21+00:00","maintenance_label":"Dormant","stars_delta_30d":0,"url":"https://www.graphcanon.com/tools/nvidia-star-attention","markdown_url":"https://www.graphcanon.com/tools/nvidia-star-attention.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-star-attention","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-star-attention"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"attention-mechanism","name":"attention-mechanism"},{"slug":"large-language-models","name":"large language models"},{"slug":"llm-inference","name":"llm-inference"}],"edges":[],"neighbours":[{"slug":"lyogavin-airllm","name":"airllm","tagline":"AirLLM 70B inference with single 4GB GPU","github_url":"https://github.com/lyogavin/airllm","owner":"lyogavin","repo":"airllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/1113905?v=4","primary_language":"Jupyter Notebook","stars":24183,"forks":2722,"topics":["chinese-llm","chinese-nlp","finetune","generative-ai","instruct-gpt","instruction-set","llama","llm","lora","open-models","open-source","open-source-models","qlora"],"archived":false,"github_pushed_at":"2026-07-23T08:29:43+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lyogavin-airllm","markdown_url":"https://www.graphcanon.com/tools/lyogavin-airllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lyogavin-airllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lyogavin-airllm","shared_categories":["inference-serving"]},{"slug":"bitsandbytes-foundation-bitsandbytes","name":"bitsandbytes","tagline":"Large language model quantization toolkit for PyTorch.","github_url":"https://github.com/bitsandbytes-foundation/bitsandbytes","owner":"bitsandbytes-foundation","repo":"bitsandbytes","owner_avatar_url":"https://avatars.githubusercontent.com/u/175231607?v=4","primary_language":"Python","stars":8385,"forks":900,"topics":["llm","machine-learning","pytorch","qlora","quantization"],"archived":false,"github_pushed_at":"2026-07-29T18:27:51+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes","markdown_url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bitsandbytes-foundation-bitsandbytes","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bitsandbytes-foundation-bitsandbytes","shared_categories":["inference-serving"]},{"slug":"eleutherai-gpt-neox","name":"gpt-neox","tagline":"Implementation of model parallel autoregressive transformers on GPUs based on Megatron and DeepSpeed libraries","github_url":"https://github.com/EleutherAI/gpt-neox","owner":"EleutherAI","repo":"gpt-neox","owner_avatar_url":"https://avatars.githubusercontent.com/u/68924597?v=4","primary_language":"Python","stars":7452,"forks":1119,"topics":["deepspeed-library","gpt-3","language-model","transformers"],"archived":false,"github_pushed_at":"2026-06-11T19:25:44+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/eleutherai-gpt-neox","markdown_url":"https://www.graphcanon.com/tools/eleutherai-gpt-neox.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eleutherai-gpt-neox","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eleutherai-gpt-neox","shared_categories":[]},{"slug":"linkedin-liger-kernel","name":"Liger-Kernel","tagline":"Efficient Triton Kernels for LLM Training","github_url":"https://github.com/linkedin/Liger-Kernel","owner":"linkedin","repo":"Liger-Kernel","owner_avatar_url":"https://avatars.githubusercontent.com/u/357098?v=4","primary_language":"Python","stars":6555,"forks":573,"topics":["finetuning","gemma2","hacktoberfest","llama","llama3","llm-training","llms","mistral","phi3","triton","triton-kernels"],"archived":false,"github_pushed_at":"2026-08-07T08:48:09+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/linkedin-liger-kernel","markdown_url":"https://www.graphcanon.com/tools/linkedin-liger-kernel.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/linkedin-liger-kernel","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=linkedin-liger-kernel","shared_categories":[]},{"slug":"nvidia-fastertransformer","name":"FasterTransformer","tagline":"Transformer related optimization including BERT and GPT","github_url":"https://github.com/NVIDIA/FasterTransformer","owner":"NVIDIA","repo":"FasterTransformer","owner_avatar_url":"https://avatars.githubusercontent.com/u/1728152?v=4","primary_language":"C++","stars":6446,"forks":935,"topics":["bert","gpt","pytorch","transformer"],"archived":false,"github_pushed_at":"2024-03-27T11:25:30+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/nvidia-fastertransformer","markdown_url":"https://www.graphcanon.com/tools/nvidia-fastertransformer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-fastertransformer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-fastertransformer","shared_categories":["inference-serving"]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["inference-serving"]},{"slug":"fla-org-flash-linear-attention","name":"flash-linear-attention","tagline":"🚀 Efficient implementations for emerging model architectures","github_url":"https://github.com/fla-org/flash-linear-attention","owner":"fla-org","repo":"flash-linear-attention","owner_avatar_url":"https://avatars.githubusercontent.com/u/40835596?v=4","primary_language":"Python","stars":5568,"forks":661,"topics":["large-language-models","machine-learning-systems","natural-language-processing","sequence-modeling"],"archived":false,"github_pushed_at":"2026-08-17T10:13:08+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/fla-org-flash-linear-attention","markdown_url":"https://www.graphcanon.com/tools/fla-org-flash-linear-attention.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fla-org-flash-linear-attention","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fla-org-flash-linear-attention","shared_categories":[]},{"slug":"xlite-dev-awesome-llm-inference","name":"Awesome-LLM-Inference","tagline":"A curated list of LLM/VLM inference papers with codes","github_url":"https://github.com/xlite-dev/Awesome-LLM-Inference","owner":"xlite-dev","repo":"Awesome-LLM-Inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/204302598?v=4","primary_language":"Python","stars":5477,"forks":429,"topics":["awesome-llm","deepseek","deepseek-r1","deepseek-v3","flash-attention","flash-attention-3","flash-mla","llm-inference","minimax-01","mla","paged-attention","qwen3","tensorrt-llm","vllm"],"archived":false,"github_pushed_at":"2026-08-14T12:23:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference","markdown_url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xlite-dev-awesome-llm-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xlite-dev-awesome-llm-inference","shared_categories":["inference-serving"]},{"slug":"turboderp-exllama","name":"exllama","tagline":"Memory-efficient rewrite of HF transformers for Llama with quantized weights","github_url":"https://github.com/turboderp/exllama","owner":"turboderp","repo":"exllama","owner_avatar_url":"https://avatars.githubusercontent.com/u/11859846?v=4","primary_language":"Python","stars":2937,"forks":220,"topics":[],"archived":false,"github_pushed_at":"2023-09-30T19:06:04+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/turboderp-exllama","markdown_url":"https://www.graphcanon.com/tools/turboderp-exllama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/turboderp-exllama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=turboderp-exllama","shared_categories":["inference-serving"]},{"slug":"fasterdecoding-medusa","name":"Medusa","tagline":"Framework for accelerating LLM generation using multiple decoding heads","github_url":"https://github.com/FasterDecoding/Medusa","owner":"FasterDecoding","repo":"Medusa","owner_avatar_url":"https://avatars.githubusercontent.com/u/144572371?v=4","primary_language":"Jupyter Notebook","stars":2767,"forks":205,"topics":["llm","llm-inference"],"archived":false,"github_pushed_at":"2024-06-25T12:23:04+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/fasterdecoding-medusa","markdown_url":"https://www.graphcanon.com/tools/fasterdecoding-medusa.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fasterdecoding-medusa","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fasterdecoding-medusa","shared_categories":["inference-serving"]},{"slug":"thudm-p-tuning-v2","name":"P-tuning-v2","tagline":"Optimized deep prompt tuning strategy comparable to fine-tuning across scales and tasks","github_url":"https://github.com/THUDM/P-tuning-v2","owner":"THUDM","repo":"P-tuning-v2","owner_avatar_url":"https://avatars.githubusercontent.com/u/48590610?v=4","primary_language":"Python","stars":2077,"forks":213,"topics":["natural-language-processing","p-tuning","parameter-efficient-learning","pretrained-language-model","prompt-tuning"],"archived":false,"github_pushed_at":"2023-11-16T04:38:09+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/thudm-p-tuning-v2","markdown_url":"https://www.graphcanon.com/tools/thudm-p-tuning-v2.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/thudm-p-tuning-v2","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=thudm-p-tuning-v2","shared_categories":[]},{"slug":"thudm-longwriter","name":"LongWriter","tagline":"LongWriter enables generation of texts longer than 10,000 words using long-context LLMs","github_url":"https://github.com/THUDM/LongWriter","owner":"THUDM","repo":"LongWriter","owner_avatar_url":"https://avatars.githubusercontent.com/u/48590610?v=4","primary_language":"Python","stars":1872,"forks":182,"topics":["fine-tuning","llm","long-context","long-text"],"archived":false,"github_pushed_at":"2025-06-24T06:41:41+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/thudm-longwriter","markdown_url":"https://www.graphcanon.com/tools/thudm-longwriter.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/thudm-longwriter","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=thudm-longwriter","shared_categories":[]}]}}