{"data":{"node":{"slug":"bitsandbytes-foundation-bitsandbytes","name":"bitsandbytes","tagline":"Large language model quantization toolkit for PyTorch.","github_url":"https://github.com/bitsandbytes-foundation/bitsandbytes","owner":"bitsandbytes-foundation","repo":"bitsandbytes","owner_avatar_url":"https://avatars.githubusercontent.com/u/175231607?v=4","primary_language":"Python","stars":8385,"forks":900,"topics":["llm","machine-learning","pytorch","qlora","quantization"],"archived":false,"github_pushed_at":"2026-07-29T18:27:51+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes","markdown_url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bitsandbytes-foundation-bitsandbytes","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bitsandbytes-foundation-bitsandbytes"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"},{"slug":"llm-frameworks","name":"LLM Frameworks","url":"https://www.graphcanon.com/categories/llm-frameworks","markdown_url":"https://www.graphcanon.com/categories/llm-frameworks.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/llm-frameworks"}],"tags":[{"slug":"llm","name":"llm"},{"slug":"machine-learning","name":"machine-learning"},{"slug":"pytorch","name":"pytorch"},{"slug":"qlora","name":"qlora"},{"slug":"quantization","name":"quantization"}],"edges":[{"type":"integrates_with","direction":"in","explanation":"Bits and bytes is a quantization toolkit specifically designed to work with PyTorch, optimizing large language models.","successor_context":null,"tool":{"slug":"pytorch-pytorch","name":"pytorch","tagline":"Tensors and Dynamic neural networks in Python with strong GPU acceleration","github_url":"https://github.com/pytorch/pytorch","owner":"pytorch","repo":"pytorch","owner_avatar_url":"https://avatars.githubusercontent.com/u/21003710?v=4","primary_language":"Python","stars":102144,"forks":28650,"topics":["autograd","deep-learning","gpu","machine-learning","neural-network","numpy","python","tensor"],"archived":false,"github_pushed_at":"2026-08-03T06:00:50+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/pytorch-pytorch","markdown_url":"https://www.graphcanon.com/tools/pytorch-pytorch.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pytorch-pytorch","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pytorch-pytorch"}}],"neighbours":[{"slug":"alexsjones-llmfit","name":"llmfit","tagline":"Hundreds of models & providers. One command to find what runs on your hardware.","github_url":"https://github.com/AlexsJones/llmfit","owner":"AlexsJones","repo":"llmfit","owner_avatar_url":"https://avatars.githubusercontent.com/u/1235925?v=4","primary_language":"Rust","stars":31867,"forks":1978,"topics":["gguf","llm","localai","mlx","skill","unsloth"],"archived":false,"github_pushed_at":"2026-08-14T07:36:41+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/alexsjones-llmfit","markdown_url":"https://www.graphcanon.com/tools/alexsjones-llmfit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alexsjones-llmfit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alexsjones-llmfit","shared_categories":["llm-frameworks"]},{"slug":"lyogavin-airllm","name":"airllm","tagline":"AirLLM 70B inference with single 4GB GPU","github_url":"https://github.com/lyogavin/airllm","owner":"lyogavin","repo":"airllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/1113905?v=4","primary_language":"Jupyter Notebook","stars":24183,"forks":2722,"topics":["chinese-llm","chinese-nlp","finetune","generative-ai","instruct-gpt","instruction-set","llama","llm","lora","open-models","open-source","open-source-models","qlora"],"archived":false,"github_pushed_at":"2026-07-23T08:29:43+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lyogavin-airllm","markdown_url":"https://www.graphcanon.com/tools/lyogavin-airllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lyogavin-airllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lyogavin-airllm","shared_categories":["inference-serving"]},{"slug":"lightning-ai-litgpt","name":"litgpt","tagline":"High-performance LLMs with recipes for pretraining, finetuning and deployment","github_url":"https://github.com/Lightning-AI/litgpt","owner":"Lightning-AI","repo":"litgpt","owner_avatar_url":"https://avatars.githubusercontent.com/u/58386951?v=4","primary_language":"Python","stars":13605,"forks":1483,"topics":["ai","artificial-intelligence","deep-learning","large-language-models","llm","llm-inference","llms"],"archived":false,"github_pushed_at":"2026-07-20T10:24:12+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lightning-ai-litgpt","markdown_url":"https://www.graphcanon.com/tools/lightning-ai-litgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lightning-ai-litgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lightning-ai-litgpt","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"artidoro-qlora","name":"qlora","tagline":"QLoRA finetuning of quantized LLMs","github_url":"https://github.com/artidoro/qlora","owner":"artidoro","repo":"qlora","owner_avatar_url":"https://avatars.githubusercontent.com/u/11949572?v=4","primary_language":"Jupyter Notebook","stars":10979,"forks":876,"topics":[],"archived":false,"github_pushed_at":"2024-06-10T19:20:16+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/artidoro-qlora","markdown_url":"https://www.graphcanon.com/tools/artidoro-qlora.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/artidoro-qlora","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=artidoro-qlora","shared_categories":["llm-frameworks"]},{"slug":"oumi-ai-oumi","name":"oumi","tagline":"Easily fine-tune, evaluate and deploy open source LLMs/VLMs","github_url":"https://github.com/oumi-ai/oumi","owner":"oumi-ai","repo":"oumi","owner_avatar_url":"https://avatars.githubusercontent.com/u/167452922?v=4","primary_language":"Python","stars":9376,"forks":784,"topics":["dpo","evaluation","fine-tuning","gpt-oss","gpt-oss-120b","gpt-oss-20b","inference","llama","llms","open-weight","open-weight-models","open-weights","sft","slms","vlms"],"archived":false,"github_pushed_at":"2026-08-21T23:11:35+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/oumi-ai-oumi","markdown_url":"https://www.graphcanon.com/tools/oumi-ai-oumi.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/oumi-ai-oumi","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=oumi-ai-oumi","shared_categories":["inference-serving"]},{"slug":"fareedkhan-dev-train-llm-from-scratch","name":"train-llm-from-scratch","tagline":"A straightforward method for training your LLM from raw text to aligned model generation","github_url":"https://github.com/FareedKhan-dev/train-llm-from-scratch","owner":"FareedKhan-dev","repo":"train-llm-from-scratch","owner_avatar_url":"https://avatars.githubusercontent.com/u/63067900?v=4","primary_language":"Python","stars":9141,"forks":1264,"topics":["gemini","large-language-models","llm","openai","training","transformers"],"archived":false,"github_pushed_at":"2026-08-17T05:07:26+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/fareedkhan-dev-train-llm-from-scratch","markdown_url":"https://www.graphcanon.com/tools/fareedkhan-dev-train-llm-from-scratch.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fareedkhan-dev-train-llm-from-scratch","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fareedkhan-dev-train-llm-from-scratch","shared_categories":["inference-serving"]},{"slug":"nvidia-fastertransformer","name":"FasterTransformer","tagline":"Transformer related optimization including BERT and GPT","github_url":"https://github.com/NVIDIA/FasterTransformer","owner":"NVIDIA","repo":"FasterTransformer","owner_avatar_url":"https://avatars.githubusercontent.com/u/1728152?v=4","primary_language":"C++","stars":6446,"forks":935,"topics":["bert","gpt","pytorch","transformer"],"archived":false,"github_pushed_at":"2024-03-27T11:25:30+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/nvidia-fastertransformer","markdown_url":"https://www.graphcanon.com/tools/nvidia-fastertransformer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-fastertransformer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-fastertransformer","shared_categories":["inference-serving"]},{"slug":"andyyyy64-whichllm","name":"whichllm","tagline":"Command-line tool to find and benchmark local LLM performance","github_url":"https://github.com/Andyyyy64/whichllm","owner":"Andyyyy64","repo":"whichllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/105579829?v=4","primary_language":"Python","stars":6225,"forks":330,"topics":["ai","apple-silicon","benchmarks","cli","command-line-tool","gguf","gpu","huggingface","inference","llm","local-llm","ollama","python","vram"],"archived":false,"github_pushed_at":"2026-08-05T07:15:32+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/andyyyy64-whichllm","markdown_url":"https://www.graphcanon.com/tools/andyyyy64-whichllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/andyyyy64-whichllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=andyyyy64-whichllm","shared_categories":["inference-serving"]},{"slug":"xlite-dev-awesome-llm-inference","name":"Awesome-LLM-Inference","tagline":"A curated list of LLM/VLM inference papers with codes","github_url":"https://github.com/xlite-dev/Awesome-LLM-Inference","owner":"xlite-dev","repo":"Awesome-LLM-Inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/204302598?v=4","primary_language":"Python","stars":5477,"forks":429,"topics":["awesome-llm","deepseek","deepseek-r1","deepseek-v3","flash-attention","flash-attention-3","flash-mla","llm-inference","minimax-01","mla","paged-attention","qwen3","tensorrt-llm","vllm"],"archived":false,"github_pushed_at":"2026-08-14T12:23:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference","markdown_url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xlite-dev-awesome-llm-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xlite-dev-awesome-llm-inference","shared_categories":["inference-serving"]},{"slug":"turboderp-exllama","name":"exllama","tagline":"Memory-efficient rewrite of HF transformers for Llama with quantized weights","github_url":"https://github.com/turboderp/exllama","owner":"turboderp","repo":"exllama","owner_avatar_url":"https://avatars.githubusercontent.com/u/11859846?v=4","primary_language":"Python","stars":2937,"forks":220,"topics":[],"archived":false,"github_pushed_at":"2023-09-30T19:06:04+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/turboderp-exllama","markdown_url":"https://www.graphcanon.com/tools/turboderp-exllama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/turboderp-exllama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=turboderp-exllama","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"huangowen-awesome-llm-compression","name":"Awesome-LLM-Compression","tagline":"Awesome LLM compression research papers and tools to accelerate LLM training and inference.","github_url":"https://github.com/HuangOwen/Awesome-LLM-Compression","owner":"HuangOwen","repo":"Awesome-LLM-Compression","owner_avatar_url":"https://avatars.githubusercontent.com/u/24937399?v=4","primary_language":null,"stars":1859,"forks":129,"topics":[],"archived":false,"github_pushed_at":"2026-06-30T15:26:46+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression","markdown_url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huangowen-awesome-llm-compression","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huangowen-awesome-llm-compression","shared_categories":["llm-frameworks","inference-serving"]},{"slug":"tensorflow-model-optimization","name":"model-optimization","tagline":"Toolkit for optimizing ML models in Keras and TensorFlow","github_url":"https://github.com/tensorflow/model-optimization","owner":"tensorflow","repo":"model-optimization","owner_avatar_url":"https://avatars.githubusercontent.com/u/15658638?v=4","primary_language":"Python","stars":1576,"forks":346,"topics":["compression","deep-learning","keras","machine-learning","ml","model-compression","optimization","pruning","quantization","quantized-networks","quantized-neural-networks","quantized-training","sparsity","tensorflow"],"archived":false,"github_pushed_at":"2026-07-27T06:03:18+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/tensorflow-model-optimization","markdown_url":"https://www.graphcanon.com/tools/tensorflow-model-optimization.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tensorflow-model-optimization","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tensorflow-model-optimization","shared_categories":[]}]}}