{"data":{"node":{"slug":"villagecomputing-superpipe","name":"superpipe","tagline":"Optimized LLM pipelines for structured data","github_url":"https://github.com/villagecomputing/superpipe","owner":"villagecomputing","repo":"superpipe","owner_avatar_url":"https://avatars.githubusercontent.com/u/159300425?v=4","primary_language":"Python","stars":109,"forks":2,"topics":["classification","data-extraction","data-labeling","llm","llm-evaluation","llm-optimization","structured-data"],"archived":false,"github_pushed_at":"2024-06-18T15:18:28+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/villagecomputing-superpipe","markdown_url":"https://www.graphcanon.com/tools/villagecomputing-superpipe.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/villagecomputing-superpipe","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=villagecomputing-superpipe"},"categories":[{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"},{"slug":"llm-frameworks","name":"LLM Frameworks","url":"https://www.graphcanon.com/categories/llm-frameworks","markdown_url":"https://www.graphcanon.com/categories/llm-frameworks.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/llm-frameworks"},{"slug":"model-training","name":"Model Training","url":"https://www.graphcanon.com/categories/model-training","markdown_url":"https://www.graphcanon.com/categories/model-training.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/model-training"}],"tags":[{"slug":"classification","name":"classification"},{"slug":"data-extraction","name":"data-extraction"},{"slug":"data-labeling","name":"data-labeling"},{"slug":"llm-optimization","name":"llm-optimization"},{"slug":"structured-data","name":"structured-data"}],"edges":[],"neighbours":[{"slug":"lightning-ai-litgpt","name":"litgpt","tagline":"High-performance LLMs with recipes for pretraining, finetuning and deployment","github_url":"https://github.com/Lightning-AI/litgpt","owner":"Lightning-AI","repo":"litgpt","owner_avatar_url":"https://avatars.githubusercontent.com/u/58386951?v=4","primary_language":"Python","stars":13605,"forks":1483,"topics":["ai","artificial-intelligence","deep-learning","large-language-models","llm","llm-inference","llms"],"archived":false,"github_pushed_at":"2026-07-20T10:24:12+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lightning-ai-litgpt","markdown_url":"https://www.graphcanon.com/tools/lightning-ai-litgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lightning-ai-litgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lightning-ai-litgpt","shared_categories":["model-training","llm-frameworks"]},{"slug":"fareedkhan-dev-train-llm-from-scratch","name":"train-llm-from-scratch","tagline":"A straightforward method for training your LLM from raw text to aligned model generation","github_url":"https://github.com/FareedKhan-dev/train-llm-from-scratch","owner":"FareedKhan-dev","repo":"train-llm-from-scratch","owner_avatar_url":"https://avatars.githubusercontent.com/u/63067900?v=4","primary_language":"Python","stars":9141,"forks":1264,"topics":["gemini","large-language-models","llm","openai","training","transformers"],"archived":false,"github_pushed_at":"2026-08-17T05:07:26+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/fareedkhan-dev-train-llm-from-scratch","markdown_url":"https://www.graphcanon.com/tools/fareedkhan-dev-train-llm-from-scratch.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fareedkhan-dev-train-llm-from-scratch","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fareedkhan-dev-train-llm-from-scratch","shared_categories":["model-training"]},{"slug":"bitsandbytes-foundation-bitsandbytes","name":"bitsandbytes","tagline":"Large language model quantization toolkit for PyTorch.","github_url":"https://github.com/bitsandbytes-foundation/bitsandbytes","owner":"bitsandbytes-foundation","repo":"bitsandbytes","owner_avatar_url":"https://avatars.githubusercontent.com/u/175231607?v=4","primary_language":"Python","stars":8385,"forks":900,"topics":["llm","machine-learning","pytorch","qlora","quantization"],"archived":false,"github_pushed_at":"2026-07-29T18:27:51+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes","markdown_url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bitsandbytes-foundation-bitsandbytes","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bitsandbytes-foundation-bitsandbytes","shared_categories":["llm-frameworks"]},{"slug":"flashinfer-ai-flashinfer","name":"flashinfer","tagline":"FlashInfer is a kernel library for serving large language models","github_url":"https://github.com/flashinfer-ai/flashinfer","owner":"flashinfer-ai","repo":"flashinfer","owner_avatar_url":"https://avatars.githubusercontent.com/u/145061914?v=4","primary_language":"Python","stars":6231,"forks":1327,"topics":["attention","cuda","distributed-inference","gpu","jit","large-large-models","llm-inference","moe","nvidia","pytorch"],"archived":false,"github_pushed_at":"2026-08-24T17:00:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer","markdown_url":"https://www.graphcanon.com/tools/flashinfer-ai-flashinfer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flashinfer-ai-flashinfer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flashinfer-ai-flashinfer","shared_categories":["llm-frameworks"]},{"slug":"xlite-dev-awesome-llm-inference","name":"Awesome-LLM-Inference","tagline":"A curated list of LLM/VLM inference papers with codes","github_url":"https://github.com/xlite-dev/Awesome-LLM-Inference","owner":"xlite-dev","repo":"Awesome-LLM-Inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/204302598?v=4","primary_language":"Python","stars":5477,"forks":429,"topics":["awesome-llm","deepseek","deepseek-r1","deepseek-v3","flash-attention","flash-attention-3","flash-mla","llm-inference","minimax-01","mla","paged-attention","qwen3","tensorrt-llm","vllm"],"archived":false,"github_pushed_at":"2026-08-14T12:23:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference","markdown_url":"https://www.graphcanon.com/tools/xlite-dev-awesome-llm-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xlite-dev-awesome-llm-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xlite-dev-awesome-llm-inference","shared_categories":[]},{"slug":"eth-sri-lmql","name":"lmql","tagline":"A language for constraint-guided and efficient LLM programming.","github_url":"https://github.com/eth-sri/lmql","owner":"eth-sri","repo":"lmql","owner_avatar_url":"https://avatars.githubusercontent.com/u/5363413?v=4","primary_language":"Python","stars":4203,"forks":221,"topics":["chatgpt","huggingface","language-model","programming-language"],"archived":false,"github_pushed_at":"2025-05-22T07:32:31+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/eth-sri-lmql","markdown_url":"https://www.graphcanon.com/tools/eth-sri-lmql.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eth-sri-lmql","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eth-sri-lmql","shared_categories":["llm-frameworks"]},{"slug":"kubeflow-pipelines","name":"pipelines","tagline":"Machine Learning Pipelines for Kubeflow","github_url":"https://github.com/kubeflow/pipelines","owner":"kubeflow","repo":"pipelines","owner_avatar_url":"https://avatars.githubusercontent.com/u/33164907?v=4","primary_language":"Python","stars":4173,"forks":2075,"topics":["data-science","kubeflow","kubeflow-pipelines","kubernetes","machine-learning","mlops","pipeline"],"archived":false,"github_pushed_at":"2026-08-03T15:35:16+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/kubeflow-pipelines","markdown_url":"https://www.graphcanon.com/tools/kubeflow-pipelines.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kubeflow-pipelines","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kubeflow-pipelines","shared_categories":["model-training"]},{"slug":"ucbepic-docetl","name":"docetl","tagline":"A system for agentic LLM-powered data processing and ETL","github_url":"https://github.com/ucbepic/docetl","owner":"ucbepic","repo":"docetl","owner_avatar_url":"https://avatars.githubusercontent.com/u/88680502?v=4","primary_language":"Python","stars":3961,"forks":421,"topics":["agents","data","data-pipelines","document-analysis","document-processing","elt","etl","llm","python","semantic-data","unstructured-data","unstructured-data-analysis","workflow"],"archived":false,"github_pushed_at":"2026-08-09T23:31:04+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ucbepic-docetl","markdown_url":"https://www.graphcanon.com/tools/ucbepic-docetl.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ucbepic-docetl","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ucbepic-docetl","shared_categories":["data-retrieval"]},{"slug":"fasterdecoding-medusa","name":"Medusa","tagline":"Framework for accelerating LLM generation using multiple decoding heads","github_url":"https://github.com/FasterDecoding/Medusa","owner":"FasterDecoding","repo":"Medusa","owner_avatar_url":"https://avatars.githubusercontent.com/u/144572371?v=4","primary_language":"Jupyter Notebook","stars":2767,"forks":205,"topics":["llm","llm-inference"],"archived":false,"github_pushed_at":"2024-06-25T12:23:04+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/fasterdecoding-medusa","markdown_url":"https://www.graphcanon.com/tools/fasterdecoding-medusa.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fasterdecoding-medusa","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fasterdecoding-medusa","shared_categories":[]},{"slug":"huangowen-awesome-llm-compression","name":"Awesome-LLM-Compression","tagline":"Awesome LLM compression research papers and tools to accelerate LLM training and inference.","github_url":"https://github.com/HuangOwen/Awesome-LLM-Compression","owner":"HuangOwen","repo":"Awesome-LLM-Compression","owner_avatar_url":"https://avatars.githubusercontent.com/u/24937399?v=4","primary_language":null,"stars":1859,"forks":129,"topics":[],"archived":false,"github_pushed_at":"2026-06-30T15:26:46+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression","markdown_url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huangowen-awesome-llm-compression","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huangowen-awesome-llm-compression","shared_categories":["llm-frameworks"]},{"slug":"nvidia-nemo-curator","name":"Curator","tagline":"Scalable data pre-processing and curation toolkit for LLMs","github_url":"https://github.com/NVIDIA-NeMo/Curator","owner":"NVIDIA-NeMo","repo":"Curator","owner_avatar_url":"https://avatars.githubusercontent.com/u/213689629?v=4","primary_language":"Python","stars":1731,"forks":320,"topics":["data","data-curation","data-prep","data-preparation","data-processing","data-processing-pipelines","data-quality","datacuration","datarecipes","deduplication","fast-data-processing","fine-tuning","large-language-models","large-scale-data-processing","llm","llm-data-quality","llmapps","python","semantic-deduplication"],"archived":false,"github_pushed_at":"2026-08-21T20:19:34+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/nvidia-nemo-curator","markdown_url":"https://www.graphcanon.com/tools/nvidia-nemo-curator.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-nemo-curator","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-nemo-curator","shared_categories":["model-training","data-retrieval"]},{"slug":"benman1-generative-ai-with-langchain","name":"generative_ai_with_langchain","tagline":"Build production-ready LLM applications and advanced agents using Python, LangChain, and LangGraph","github_url":"https://github.com/benman1/generative_ai_with_langchain","owner":"benman1","repo":"generative_ai_with_langchain","owner_avatar_url":"https://avatars.githubusercontent.com/u/10786684?v=4","primary_language":"Jupyter Notebook","stars":1400,"forks":582,"topics":["agent","chatgpt","claude","claude-3-5-sonnet","deepseek","deepseek-r1","gpt","gpt-4o","huggingface","langchain","langgraph","llamacpp","llms","ollama","openai"],"archived":false,"github_pushed_at":"2026-08-05T12:50:30+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/benman1-generative-ai-with-langchain","markdown_url":"https://www.graphcanon.com/tools/benman1-generative-ai-with-langchain.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/benman1-generative-ai-with-langchain","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=benman1-generative-ai-with-langchain","shared_categories":["llm-frameworks"]}]}}