{"data":{"node":{"slug":"huggingface-tokenizers","name":"tokenizers","tagline":"💥 Fast State-of-the-Art Tokenizers optimized for Research and Production","github_url":"https://github.com/huggingface/tokenizers","owner":"huggingface","repo":"tokenizers","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Rust","stars":10940,"forks":1160,"topics":["bert","gpt","language-model","natural-language-processing","natural-language-understanding","nlp","transformers"],"archived":false,"github_pushed_at":"2026-08-01T12:35:36+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/huggingface-tokenizers","markdown_url":"https://www.graphcanon.com/tools/huggingface-tokenizers.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-tokenizers","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-tokenizers"},"categories":[{"slug":"llm-frameworks","name":"LLM Frameworks","url":"https://www.graphcanon.com/categories/llm-frameworks","markdown_url":"https://www.graphcanon.com/categories/llm-frameworks.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/llm-frameworks"},{"slug":"model-training","name":"Model Training","url":"https://www.graphcanon.com/categories/model-training","markdown_url":"https://www.graphcanon.com/categories/model-training.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/model-training"}],"tags":[{"slug":"bert","name":"bert"},{"slug":"gpt","name":"gpt"},{"slug":"language-model","name":"language-model"},{"slug":"natural-language-processing","name":"natural-language-processing"},{"slug":"natural-language-understanding","name":"natural-language-understanding"},{"slug":"nlp","name":"nlp"},{"slug":"transformers","name":"transformers"}],"edges":[],"neighbours":[{"slug":"lightning-ai-litgpt","name":"litgpt","tagline":"High-performance LLMs with recipes for pretraining, finetuning and deployment","github_url":"https://github.com/Lightning-AI/litgpt","owner":"Lightning-AI","repo":"litgpt","owner_avatar_url":"https://avatars.githubusercontent.com/u/58386951?v=4","primary_language":"Python","stars":13605,"forks":1483,"topics":["ai","artificial-intelligence","deep-learning","large-language-models","llm","llm-inference","llms"],"archived":false,"github_pushed_at":"2026-07-20T10:24:12+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lightning-ai-litgpt","markdown_url":"https://www.graphcanon.com/tools/lightning-ai-litgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lightning-ai-litgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lightning-ai-litgpt","shared_categories":["model-training","llm-frameworks"]},{"slug":"fareedkhan-dev-train-llm-from-scratch","name":"train-llm-from-scratch","tagline":"A straightforward method for training your LLM from raw text to aligned model generation","github_url":"https://github.com/FareedKhan-dev/train-llm-from-scratch","owner":"FareedKhan-dev","repo":"train-llm-from-scratch","owner_avatar_url":"https://avatars.githubusercontent.com/u/63067900?v=4","primary_language":"Python","stars":9141,"forks":1264,"topics":["gemini","large-language-models","llm","openai","training","transformers"],"archived":false,"github_pushed_at":"2026-08-17T05:07:26+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/fareedkhan-dev-train-llm-from-scratch","markdown_url":"https://www.graphcanon.com/tools/fareedkhan-dev-train-llm-from-scratch.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fareedkhan-dev-train-llm-from-scratch","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fareedkhan-dev-train-llm-from-scratch","shared_categories":["model-training"]},{"slug":"bitsandbytes-foundation-bitsandbytes","name":"bitsandbytes","tagline":"Large language model quantization toolkit for PyTorch.","github_url":"https://github.com/bitsandbytes-foundation/bitsandbytes","owner":"bitsandbytes-foundation","repo":"bitsandbytes","owner_avatar_url":"https://avatars.githubusercontent.com/u/175231607?v=4","primary_language":"Python","stars":8385,"forks":900,"topics":["llm","machine-learning","pytorch","qlora","quantization"],"archived":false,"github_pushed_at":"2026-07-29T18:27:51+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes","markdown_url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bitsandbytes-foundation-bitsandbytes","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bitsandbytes-foundation-bitsandbytes","shared_categories":["llm-frameworks"]},{"slug":"eleutherai-gpt-neox","name":"gpt-neox","tagline":"Implementation of model parallel autoregressive transformers on GPUs based on Megatron and DeepSpeed libraries","github_url":"https://github.com/EleutherAI/gpt-neox","owner":"EleutherAI","repo":"gpt-neox","owner_avatar_url":"https://avatars.githubusercontent.com/u/68924597?v=4","primary_language":"Python","stars":7452,"forks":1119,"topics":["deepspeed-library","gpt-3","language-model","transformers"],"archived":false,"github_pushed_at":"2026-06-11T19:25:44+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/eleutherai-gpt-neox","markdown_url":"https://www.graphcanon.com/tools/eleutherai-gpt-neox.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eleutherai-gpt-neox","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eleutherai-gpt-neox","shared_categories":["model-training","llm-frameworks"]},{"slug":"nvidia-fastertransformer","name":"FasterTransformer","tagline":"Transformer related optimization including BERT and GPT","github_url":"https://github.com/NVIDIA/FasterTransformer","owner":"NVIDIA","repo":"FasterTransformer","owner_avatar_url":"https://avatars.githubusercontent.com/u/1728152?v=4","primary_language":"C++","stars":6446,"forks":935,"topics":["bert","gpt","pytorch","transformer"],"archived":false,"github_pushed_at":"2024-03-27T11:25:30+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/nvidia-fastertransformer","markdown_url":"https://www.graphcanon.com/tools/nvidia-fastertransformer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-fastertransformer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-fastertransformer","shared_categories":[]},{"slug":"sanchit-gandhi-whisper-jax","name":"whisper-jax","tagline":"JAX implementation of OpenAI's Whisper model for up to 70x speed-up on TPU.","github_url":"https://github.com/sanchit-gandhi/whisper-jax","owner":"sanchit-gandhi","repo":"whisper-jax","owner_avatar_url":"https://avatars.githubusercontent.com/u/93869735?v=4","primary_language":"Jupyter Notebook","stars":4684,"forks":411,"topics":["deep-learning","jax","speech-recognition","speech-to-text","whisper"],"archived":false,"github_pushed_at":"2024-04-03T12:12:52+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/sanchit-gandhi-whisper-jax","markdown_url":"https://www.graphcanon.com/tools/sanchit-gandhi-whisper-jax.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/sanchit-gandhi-whisper-jax","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=sanchit-gandhi-whisper-jax","shared_categories":[]},{"slug":"qdrant-fastembed","name":"fastembed","tagline":"Fast, Accurate, Lightweight Python library for creating state-of-the-art embeddings","github_url":"https://github.com/qdrant/fastembed","owner":"qdrant","repo":"fastembed","owner_avatar_url":"https://avatars.githubusercontent.com/u/73504361?v=4","primary_language":"Python","stars":3158,"forks":231,"topics":["embeddings","openai","rag","retrieval","retrieval-augmented-generation","vector-search"],"archived":false,"github_pushed_at":"2026-08-19T16:03:23+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/qdrant-fastembed","markdown_url":"https://www.graphcanon.com/tools/qdrant-fastembed.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/qdrant-fastembed","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=qdrant-fastembed","shared_categories":[]},{"slug":"huangowen-awesome-llm-compression","name":"Awesome-LLM-Compression","tagline":"Awesome LLM compression research papers and tools to accelerate LLM training and inference.","github_url":"https://github.com/HuangOwen/Awesome-LLM-Compression","owner":"HuangOwen","repo":"Awesome-LLM-Compression","owner_avatar_url":"https://avatars.githubusercontent.com/u/24937399?v=4","primary_language":null,"stars":1859,"forks":129,"topics":[],"archived":false,"github_pushed_at":"2026-06-30T15:26:46+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression","markdown_url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huangowen-awesome-llm-compression","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huangowen-awesome-llm-compression","shared_categories":["llm-frameworks"]},{"slug":"arahim3-mlx-tune","name":"mlx-tune","tagline":"Fine-tune LLMs on your Mac with Apple Silicon for various tasks including SFT, DPO, GRPO, Vision, TTS, STT, Embedding, and OCR.","github_url":"https://github.com/ARahim3/mlx-tune","owner":"ARahim3","repo":"mlx-tune","owner_avatar_url":"https://avatars.githubusercontent.com/u/41390319?v=4","primary_language":"Python","stars":1372,"forks":88,"topics":["apple-silicon","deep-learning","huggingface","large-language-models","llm","llm-finetuning","local-llm","lora","machine-learning","macos","mlx","on-device-ai","peft","speech-recognition","speech-to-text","text-to-speech","transformers","unsloth","vision-language-model","whisper"],"archived":false,"github_pushed_at":"2026-06-23T12:24:30+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/arahim3-mlx-tune","markdown_url":"https://www.graphcanon.com/tools/arahim3-mlx-tune.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/arahim3-mlx-tune","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=arahim3-mlx-tune","shared_categories":["model-training","llm-frameworks"]},{"slug":"jishengpeng-wavtokenizer","name":"WavTokenizer","tagline":"[ICLR 2025] State-of-the-art discrete acoustic codec models for audio language modeling","github_url":"https://github.com/jishengpeng/WavTokenizer","owner":"jishengpeng","repo":"WavTokenizer","owner_avatar_url":"https://avatars.githubusercontent.com/u/78149477?v=4","primary_language":"Python","stars":1310,"forks":113,"topics":["acoustic","audio-representation","codec","dac","encodec","gpt4o","music-representation-learning","semantic","soundstream","speech-language-model","speech-representation","text-to-speech"],"archived":false,"github_pushed_at":"2025-03-02T03:53:58+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/jishengpeng-wavtokenizer","markdown_url":"https://www.graphcanon.com/tools/jishengpeng-wavtokenizer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jishengpeng-wavtokenizer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jishengpeng-wavtokenizer","shared_categories":[]},{"slug":"bheinzerling-bpemb","name":"bpemb","tagline":"Pre-trained subword embeddings in 275 languages using Byte-Pair Encoding","github_url":"https://github.com/bheinzerling/bpemb","owner":"bheinzerling","repo":"bpemb","owner_avatar_url":"https://avatars.githubusercontent.com/u/4348795?v=4","primary_language":"Python","stars":1224,"forks":100,"topics":["embeddings","multilingual","natural-language-processing","nlp","subword-embeddings"],"archived":false,"github_pushed_at":"2024-10-01T02:49:47+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/bheinzerling-bpemb","markdown_url":"https://www.graphcanon.com/tools/bheinzerling-bpemb.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bheinzerling-bpemb","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bheinzerling-bpemb","shared_categories":[]},{"slug":"tencent-tencentpretrain","name":"TencentPretrain","tagline":"Tencent Pre-training framework in PyTorch & Pre-trained Model Zoo","github_url":"https://github.com/Tencent/TencentPretrain","owner":"Tencent","repo":"TencentPretrain","owner_avatar_url":"https://avatars.githubusercontent.com/u/18461506?v=4","primary_language":"Python","stars":1091,"forks":148,"topics":["albert","bart","bert","chinese","classification","clue","elmo","fine-tuning","gpt","gpt-2","model-zoo","natural-language-processing","ner","pegasus","pre-training","pytorch","roberta","t5","unilm","xlm-roberta"],"archived":false,"github_pushed_at":"2024-08-04T11:53:43+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/tencent-tencentpretrain","markdown_url":"https://www.graphcanon.com/tools/tencent-tencentpretrain.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tencent-tencentpretrain","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tencent-tencentpretrain","shared_categories":["model-training","llm-frameworks"]}]}}