{"data":{"slug":"xlite-dev-awesome-llm-inference","name":"Awesome-LLM-Inference","tagline":"A curated list of LLM/VLM inference papers with codes","github_url":"https://github.com/xlite-dev/Awesome-LLM-Inference","owner":"xlite-dev","repo":"Awesome-LLM-Inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/204302598?v=4","primary_language":"Python","stars":5477,"forks":429,"topics":["awesome-llm","deepseek","deepseek-r1","deepseek-v3","flash-attention","flash-attention-3","flash-mla","llm-inference","minimax-01","mla","paged-attention","qwen3","tensorrt-llm","vllm"],"archived":false,"github_pushed_at":"2026-08-14T12:23:49+00:00","maintenance_label":"Active","stars_delta_30d":62,"url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference","markdown_url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xlite-dev-awesome-llm-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xlite-dev-awesome-llm-inference","description":"📚A curated list of Awesome LLM/VLM Inference Papers with Codes: Flash-Attention, Paged-Attention, WINT8/4, Parallelism, etc.🎉","homepage_url":null,"license":"GPL-3.0","open_issues":6,"watchers":136,"ai_summary":"Gathers information on various techniques for efficient large language model and vision-language model inference like Flash-Attention, Paged-Attention.","readme_excerpt":"## ©️License\n\nGNU General Public License v3.0","github_created_at":"2023-08-27T02:32:15+00:00","created_at":"2026-07-11T11:42:46.529841+00:00","updated_at":"2026-08-24T18:01:17.66406+00:00","categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"flash-attention","name":"flash-attention"},{"slug":"paged-attention","name":"paged-attention"},{"slug":"parallelism","name":"parallelism"},{"slug":"wint8-4","name":"wint8/4"}],"trust":{"provenance":{"is_fork":false,"github_id":683573931,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-08-24T18:01:16.759Z","maintenance":{"label":"Active","score":82,"methodology":"github_public_v1","releases_90d":0,"days_since_push":10,"last_release_at":"2025-06-17T09:57:32Z","stars_delta_30d":62,"open_issues_delta_30d":0},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-11T11:42:47.630Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-08-24T18:01:17.292Z"},"languages":{"value":["python"],"source":"github.language","observed_at":"2026-08-24T18:01:17.292Z"},"license_spdx":{"value":"GPL-3.0","source":"github.license","observed_at":"2026-08-24T18:01:17.292Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":{"notes":["Requires Python for the use of included codes and to understand the methods described in the associated papers."]},"constraints":null,"when_to_use":["Use Awesome-LLM-Inference when you are looking to optimize the performance of your large language model or vision-language model inference with cutting-edge techniques such as Flash-Attention.","If you need a comprehensive list of resources for efficient inference strategies that include both recent and foundational papers, this repository is a valuable resource."],"when_not_to_use":["Do not use Awesome-LLM-Inference if your project strictly conforms to licenses different from GPL-3.0, as its licensing could be incompatible with your project's license requirements.","Avoid using this tool for immediate production implementation of inference techniques without additional vetting since the repository itself may contain unvetted research papers and code snippets."],"source":"enrich:decision_facts","observed_at":"2026-07-14T18:19:58.312Z"},"constraint_facets":null,"decision_summary":[{"label":"Requirements","value":"Requires Python for the use of included codes and to understand the methods described in the associated papers."},{"label":"Adopt for","value":"Awesome-LLM-Inference is a well-curated list of papers and codes related to efficient inference techniques for large language models and vision-language models, featuring methods like Flash-Attention and Paged-Attention."},{"label":"License detail","value":"The tool is licensed under GPL-3.0, which may affect how it can be integrated into other projects depending on their licensing needs."}]}}