{"data":{"node":{"slug":"bheinzerling-bpemb","name":"bpemb","tagline":"Pre-trained subword embeddings in 275 languages using Byte-Pair Encoding","github_url":"https://github.com/bheinzerling/bpemb","owner":"bheinzerling","repo":"bpemb","owner_avatar_url":"https://avatars.githubusercontent.com/u/4348795?v=4","primary_language":"Python","stars":1224,"forks":100,"topics":["embeddings","multilingual","natural-language-processing","nlp","subword-embeddings"],"archived":false,"github_pushed_at":"2024-10-01T02:49:47+00:00","maintenance_label":"Dormant","stars_delta_30d":2,"url":"https://www.graphcanon.com/tools/bheinzerling-bpemb","markdown_url":"https://www.graphcanon.com/tools/bheinzerling-bpemb.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bheinzerling-bpemb","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bheinzerling-bpemb"},"categories":[{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"}],"tags":[{"slug":"embeddings","name":"embeddings"},{"slug":"multilingual","name":"multilingual"},{"slug":"natural-language-processing","name":"natural-language-processing"},{"slug":"nlp","name":"nlp"},{"slug":"subword-embeddings","name":"subword-embeddings"}],"edges":[],"neighbours":[{"slug":"eleutherai-lm-evaluation-harness","name":"lm-evaluation-harness","tagline":"A framework for few-shot evaluation of language models.","github_url":"https://github.com/EleutherAI/lm-evaluation-harness","owner":"EleutherAI","repo":"lm-evaluation-harness","owner_avatar_url":"https://avatars.githubusercontent.com/u/68924597?v=4","primary_language":"Python","stars":13560,"forks":3467,"topics":["evaluation-framework","language-model","transformer"],"archived":false,"github_pushed_at":"2026-07-13T20:18:15+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness","markdown_url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eleutherai-lm-evaluation-harness","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eleutherai-lm-evaluation-harness","shared_categories":[]},{"slug":"eugeneyan-open-llms","name":"open-llms","tagline":"A list of open LLMs available for commercial use.","github_url":"https://github.com/eugeneyan/open-llms","owner":"eugeneyan","repo":"open-llms","owner_avatar_url":"https://avatars.githubusercontent.com/u/6831355?v=4","primary_language":null,"stars":12849,"forks":985,"topics":["commercial","large-language-models","llm","llms"],"archived":false,"github_pushed_at":"2025-02-13T06:37:12+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/eugeneyan-open-llms","markdown_url":"https://www.graphcanon.com/tools/eugeneyan-open-llms.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eugeneyan-open-llms","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eugeneyan-open-llms","shared_categories":[]},{"slug":"huggingface-tokenizers","name":"tokenizers","tagline":"💥 Fast State-of-the-Art Tokenizers optimized for Research and Production","github_url":"https://github.com/huggingface/tokenizers","owner":"huggingface","repo":"tokenizers","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Rust","stars":10940,"forks":1160,"topics":["bert","gpt","language-model","natural-language-processing","natural-language-understanding","nlp","transformers"],"archived":false,"github_pushed_at":"2026-08-01T12:35:36+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-tokenizers","markdown_url":"https://www.graphcanon.com/tools/huggingface-tokenizers.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-tokenizers","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-tokenizers","shared_categories":[]},{"slug":"bitsandbytes-foundation-bitsandbytes","name":"bitsandbytes","tagline":"Large language model quantization toolkit for PyTorch.","github_url":"https://github.com/bitsandbytes-foundation/bitsandbytes","owner":"bitsandbytes-foundation","repo":"bitsandbytes","owner_avatar_url":"https://avatars.githubusercontent.com/u/175231607?v=4","primary_language":"Python","stars":8385,"forks":900,"topics":["llm","machine-learning","pytorch","qlora","quantization"],"archived":false,"github_pushed_at":"2026-07-29T18:27:51+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes","markdown_url":"https://www.graphcanon.com/tools/bitsandbytes-foundation-bitsandbytes.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bitsandbytes-foundation-bitsandbytes","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bitsandbytes-foundation-bitsandbytes","shared_categories":[]},{"slug":"huggingface-text-embeddings-inference","name":"text-embeddings-inference","tagline":"Blazing fast inference solution for text embeddings models","github_url":"https://github.com/huggingface/text-embeddings-inference","owner":"huggingface","repo":"text-embeddings-inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Rust","stars":4982,"forks":421,"topics":["ai","embeddings","huggingface","llm","ml"],"archived":false,"github_pushed_at":"2026-07-24T13:47:50+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huggingface-text-embeddings-inference","markdown_url":"https://www.graphcanon.com/tools/huggingface-text-embeddings-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-text-embeddings-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-text-embeddings-inference","shared_categories":[]},{"slug":"embeddings-benchmark-mteb","name":"mteb","tagline":"State-of-the-art evaluation of embeddings across languages and modalities","github_url":"https://github.com/embeddings-benchmark/mteb","owner":"embeddings-benchmark","repo":"mteb","owner_avatar_url":"https://avatars.githubusercontent.com/u/103029531?v=4","primary_language":"Python","stars":3400,"forks":670,"topics":["benchmark","bitext-mining","clustering","embeddings","evaluation","information-retrieval","low-resource-nlp","mteb","multilingual-nlp","multimodal","neural-search","reranking","retrieval","sbert","semantic-search","sentence-transformers","sts","text-classification","text-embedding"],"archived":false,"github_pushed_at":"2026-08-21T21:26:00+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/embeddings-benchmark-mteb","markdown_url":"https://www.graphcanon.com/tools/embeddings-benchmark-mteb.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/embeddings-benchmark-mteb","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=embeddings-benchmark-mteb","shared_categories":[]},{"slug":"qdrant-fastembed","name":"fastembed","tagline":"Fast, Accurate, Lightweight Python library for creating state-of-the-art embeddings","github_url":"https://github.com/qdrant/fastembed","owner":"qdrant","repo":"fastembed","owner_avatar_url":"https://avatars.githubusercontent.com/u/73504361?v=4","primary_language":"Python","stars":3158,"forks":231,"topics":["embeddings","openai","rag","retrieval","retrieval-augmented-generation","vector-search"],"archived":false,"github_pushed_at":"2026-08-19T16:03:23+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/qdrant-fastembed","markdown_url":"https://www.graphcanon.com/tools/qdrant-fastembed.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/qdrant-fastembed","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=qdrant-fastembed","shared_categories":["data-retrieval"]},{"slug":"minishlab-model2vec","name":"model2vec","tagline":"Fast State-of-the-Art Static Embeddings","github_url":"https://github.com/MinishLab/model2vec","owner":"MinishLab","repo":"model2vec","owner_avatar_url":"https://avatars.githubusercontent.com/u/177965497?v=4","primary_language":"Python","stars":2183,"forks":123,"topics":["ai","embeddings","machine-learning","model2vec","nlp","python","sentence-transformers","word-embeddings"],"archived":false,"github_pushed_at":"2026-08-20T13:49:55+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/minishlab-model2vec","markdown_url":"https://www.graphcanon.com/tools/minishlab-model2vec.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/minishlab-model2vec","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=minishlab-model2vec","shared_categories":["data-retrieval"]},{"slug":"huangowen-awesome-llm-compression","name":"Awesome-LLM-Compression","tagline":"Awesome LLM compression research papers and tools to accelerate LLM training and inference.","github_url":"https://github.com/HuangOwen/Awesome-LLM-Compression","owner":"HuangOwen","repo":"Awesome-LLM-Compression","owner_avatar_url":"https://avatars.githubusercontent.com/u/24937399?v=4","primary_language":null,"stars":1859,"forks":129,"topics":[],"archived":false,"github_pushed_at":"2026-06-30T15:26:46+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression","markdown_url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huangowen-awesome-llm-compression","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huangowen-awesome-llm-compression","shared_categories":[]},{"slug":"hironsan-awesome-embedding-models","name":"awesome-embedding-models","tagline":"A curated list of embedding models tutorials, projects and communities.","github_url":"https://github.com/Hironsan/awesome-embedding-models","owner":"Hironsan","repo":"awesome-embedding-models","owner_avatar_url":"https://avatars.githubusercontent.com/u/6737785?v=4","primary_language":"Jupyter Notebook","stars":1850,"forks":249,"topics":["awesome","embedding-models","embeddings","machine-learning","natural-language-processing","papers","word2vec"],"archived":false,"github_pushed_at":"2019-04-07T22:56:01+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/hironsan-awesome-embedding-models","markdown_url":"https://www.graphcanon.com/tools/hironsan-awesome-embedding-models.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/hironsan-awesome-embedding-models","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=hironsan-awesome-embedding-models","shared_categories":["data-retrieval"]},{"slug":"llm-jp-awesome-japanese-llm","name":"awesome-japanese-llm","tagline":"Overview of Japanese LLMs","github_url":"https://github.com/llm-jp/awesome-japanese-llm","owner":"llm-jp","repo":"awesome-japanese-llm","owner_avatar_url":"https://avatars.githubusercontent.com/u/134031702?v=4","primary_language":"TypeScript","stars":1424,"forks":45,"topics":["foundation-models","generative-ai","generative-model","generative-models","japanese","japanese-language","japanese-language-model","japanese-llm","language-model","language-models","large-language-model","large-language-models","llm","llm-japanese","llms","multimodal","vision-and-language","vision-language","vision-language-model"],"archived":false,"github_pushed_at":"2026-08-05T12:39:04+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/llm-jp-awesome-japanese-llm","markdown_url":"https://www.graphcanon.com/tools/llm-jp-awesome-japanese-llm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/llm-jp-awesome-japanese-llm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=llm-jp-awesome-japanese-llm","shared_categories":[]},{"slug":"coqui-ai-open-speech-corpora","name":"open-speech-corpora","tagline":"A list of accessible speech corpora for ASR, TTS, and other Speech Technologies","github_url":"https://github.com/coqui-ai/open-speech-corpora","owner":"coqui-ai","repo":"open-speech-corpora","owner_avatar_url":"https://avatars.githubusercontent.com/u/75583352?v=4","primary_language":null,"stars":1398,"forks":151,"topics":["speech-emotion-recognition","speech-processing","speech-recognition","speech-separation","speech-synthesis","speech-to-text","stt","text-to-speech","tts","voice-activity-detection","voice-cloning","voice-recognition"],"archived":false,"github_pushed_at":"2024-06-06T11:33:44+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/coqui-ai-open-speech-corpora","markdown_url":"https://www.graphcanon.com/tools/coqui-ai-open-speech-corpora.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/coqui-ai-open-speech-corpora","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=coqui-ai-open-speech-corpora","shared_categories":[]}]}}