{"data":{"node":{"slug":"andrewkchan-yalm","name":"yalm","tagline":"LLM inference engine in C++/CUDA without dependency on external libraries except for I/O","github_url":"https://github.com/andrewkchan/yalm","owner":"andrewkchan","repo":"yalm","owner_avatar_url":"https://avatars.githubusercontent.com/u/8591901?v=4","primary_language":"C++","stars":596,"forks":64,"topics":["cpp","cuda","inference-engine","llama","llamacpp","llm","llm-inference","machine-learning","mistral"],"archived":false,"github_pushed_at":"2025-09-13T09:22:40+00:00","maintenance_label":"Slowing","stars_delta_30d":4,"url":"https://www.graphcanon.com/tools/andrewkchan-yalm","markdown_url":"https://www.graphcanon.com/tools/andrewkchan-yalm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/andrewkchan-yalm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=andrewkchan-yalm"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"cpp","name":"cpp"},{"slug":"cuda","name":"cuda"},{"slug":"llm-inference","name":"llm-inference"},{"slug":"machine-learning","name":"machine-learning"}],"edges":[{"type":"alternative","direction":"in","explanation":"YALM focuses on LLM inference in C++/CUDA, similar to whisper.cpp which also focuses on efficient execution of models with C/C++.","successor_context":null,"tool":{"slug":"ggml-org-whisper-cpp","name":"whisper.cpp","tagline":"Port of OpenAI's Whisper model in C/C++ for speech-to-text inference","github_url":"https://github.com/ggml-org/whisper.cpp","owner":"ggml-org","repo":"whisper.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/134263123?v=4","primary_language":"C++","stars":52501,"forks":5971,"topics":["inference","openai","speech-recognition","speech-to-text","transformer","whisper"],"archived":false,"github_pushed_at":"2026-07-31T07:11:28+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/ggml-org-whisper-cpp","markdown_url":"https://www.graphcanon.com/tools/ggml-org-whisper-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ggml-org-whisper-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ggml-org-whisper-cpp"}}],"neighbours":[{"slug":"ggml-org-llama-cpp","name":"llama.cpp","tagline":"LLM inference in C/C++","github_url":"https://github.com/ggml-org/llama.cpp","owner":"ggml-org","repo":"llama.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/134263123?v=4","primary_language":"C++","stars":122941,"forks":21406,"topics":["ggml"],"archived":false,"github_pushed_at":"2026-08-07T05:28:54+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp","markdown_url":"https://www.graphcanon.com/tools/ggml-org-llama-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ggml-org-llama-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ggml-org-llama-cpp","shared_categories":["inference-serving"]},{"slug":"lyogavin-airllm","name":"airllm","tagline":"AirLLM 70B inference with single 4GB GPU","github_url":"https://github.com/lyogavin/airllm","owner":"lyogavin","repo":"airllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/1113905?v=4","primary_language":"Jupyter Notebook","stars":24183,"forks":2722,"topics":["chinese-llm","chinese-nlp","finetune","generative-ai","instruct-gpt","instruction-set","llama","llm","lora","open-models","open-source","open-source-models","qlora"],"archived":false,"github_pushed_at":"2026-07-23T08:29:43+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lyogavin-airllm","markdown_url":"https://www.graphcanon.com/tools/lyogavin-airllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lyogavin-airllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lyogavin-airllm","shared_categories":["inference-serving"]},{"slug":"lightning-ai-litgpt","name":"litgpt","tagline":"High-performance LLMs with recipes for pretraining, finetuning and deployment","github_url":"https://github.com/Lightning-AI/litgpt","owner":"Lightning-AI","repo":"litgpt","owner_avatar_url":"https://avatars.githubusercontent.com/u/58386951?v=4","primary_language":"Python","stars":13605,"forks":1483,"topics":["ai","artificial-intelligence","deep-learning","large-language-models","llm","llm-inference","llms"],"archived":false,"github_pushed_at":"2026-07-20T10:24:12+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lightning-ai-litgpt","markdown_url":"https://www.graphcanon.com/tools/lightning-ai-litgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lightning-ai-litgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lightning-ai-litgpt","shared_categories":["inference-serving"]},{"slug":"fareedkhan-dev-train-llm-from-scratch","name":"train-llm-from-scratch","tagline":"A straightforward method for training your LLM from raw text to aligned model generation","github_url":"https://github.com/FareedKhan-dev/train-llm-from-scratch","owner":"FareedKhan-dev","repo":"train-llm-from-scratch","owner_avatar_url":"https://avatars.githubusercontent.com/u/63067900?v=4","primary_language":"Python","stars":9141,"forks":1264,"topics":["gemini","large-language-models","llm","openai","training","transformers"],"archived":false,"github_pushed_at":"2026-08-17T05:07:26+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/fareedkhan-dev-train-llm-from-scratch","markdown_url":"https://www.graphcanon.com/tools/fareedkhan-dev-train-llm-from-scratch.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fareedkhan-dev-train-llm-from-scratch","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fareedkhan-dev-train-llm-from-scratch","shared_categories":["inference-serving"]},{"slug":"bitsandbytes-foundation-bitsandbytes","name":"bitsandbytes","tagline":"Large language model quantization toolkit for PyTorch.","github_url":"https://github.com/bitsandbytes-foundation/bitsandbytes","owner":"bitsandbytes-foundation","repo":"bitsandbytes","owner_avatar_url":"https://avatars.githubusercontent.com/u/175231607?v=4","primary_language":"Python","stars":8385,"forks":900,"topics":["llm","machine-learning","pytorch","qlora","quantization"],"archived":false,"github_pushed_at":"2026-07-29T18:27:51+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes","markdown_url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bitsandbytes-foundation-bitsandbytes","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bitsandbytes-foundation-bitsandbytes","shared_categories":["inference-serving"]},{"slug":"ericlbuehler-mistral-rs","name":"mistral.rs","tagline":"Fast flexible LLM inference","github_url":"https://github.com/EricLBuehler/mistral.rs","owner":"EricLBuehler","repo":"mistral.rs","owner_avatar_url":"https://avatars.githubusercontent.com/u/65165915?v=4","primary_language":"Rust","stars":7575,"forks":671,"topics":["llm","rust","uqff"],"archived":false,"github_pushed_at":"2026-07-29T20:21:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs","markdown_url":"https://www.graphcanon.com/tools/ericlbuehler-mistral-rs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ericlbuehler-mistral-rs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ericlbuehler-mistral-rs","shared_categories":["inference-serving"]},{"slug":"eleutherai-gpt-neox","name":"gpt-neox","tagline":"Implementation of model parallel autoregressive transformers on GPUs based on Megatron and DeepSpeed libraries","github_url":"https://github.com/EleutherAI/gpt-neox","owner":"EleutherAI","repo":"gpt-neox","owner_avatar_url":"https://avatars.githubusercontent.com/u/68924597?v=4","primary_language":"Python","stars":7452,"forks":1119,"topics":["deepspeed-library","gpt-3","language-model","transformers"],"archived":false,"github_pushed_at":"2026-06-11T19:25:44+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/eleutherai-gpt-neox","markdown_url":"https://www.graphcanon.com/tools/eleutherai-gpt-neox.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eleutherai-gpt-neox","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eleutherai-gpt-neox","shared_categories":[]},{"slug":"nvidia-fastertransformer","name":"FasterTransformer","tagline":"Transformer related optimization including BERT and GPT","github_url":"https://github.com/NVIDIA/FasterTransformer","owner":"NVIDIA","repo":"FasterTransformer","owner_avatar_url":"https://avatars.githubusercontent.com/u/1728152?v=4","primary_language":"C++","stars":6446,"forks":935,"topics":["bert","gpt","pytorch","transformer"],"archived":false,"github_pushed_at":"2024-03-27T11:25:30+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/nvidia-fastertransformer","markdown_url":"https://www.graphcanon.com/tools/nvidia-fastertransformer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-fastertransformer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-fastertransformer","shared_categories":["inference-serving"]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["inference-serving"]},{"slug":"madcowd-ell","name":"ell","tagline":"A language model programming library","github_url":"https://github.com/MadcowD/ell","owner":"MadcowD","repo":"ell","owner_avatar_url":"https://avatars.githubusercontent.com/u/719535?v=4","primary_language":"Python","stars":5869,"forks":343,"topics":["ai","prompt-engineering"],"archived":false,"github_pushed_at":"2025-06-05T21:41:57+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/madcowd-ell","markdown_url":"https://www.graphcanon.com/tools/madcowd-ell.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/madcowd-ell","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=madcowd-ell","shared_categories":[]},{"slug":"xlite-dev-awesome-llm-inference","name":"Awesome-LLM-Inference","tagline":"A curated list of LLM/VLM inference papers with codes","github_url":"https://github.com/xlite-dev/Awesome-LLM-Inference","owner":"xlite-dev","repo":"Awesome-LLM-Inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/204302598?v=4","primary_language":"Python","stars":5477,"forks":429,"topics":["awesome-llm","deepseek","deepseek-r1","deepseek-v3","flash-attention","flash-attention-3","flash-mla","llm-inference","minimax-01","mla","paged-attention","qwen3","tensorrt-llm","vllm"],"archived":false,"github_pushed_at":"2026-08-14T12:23:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference","markdown_url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xlite-dev-awesome-llm-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xlite-dev-awesome-llm-inference","shared_categories":["inference-serving"]},{"slug":"turboderp-exllama","name":"exllama","tagline":"Memory-efficient rewrite of HF transformers for Llama with quantized weights","github_url":"https://github.com/turboderp/exllama","owner":"turboderp","repo":"exllama","owner_avatar_url":"https://avatars.githubusercontent.com/u/11859846?v=4","primary_language":"Python","stars":2937,"forks":220,"topics":[],"archived":false,"github_pushed_at":"2023-09-30T19:06:04+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/turboderp-exllama","markdown_url":"https://www.graphcanon.com/tools/turboderp-exllama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/turboderp-exllama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=turboderp-exllama","shared_categories":["inference-serving"]}]}}