{"data":{"node":{"slug":"michaelfeil-infinity","name":"infinity","tagline":"High-throughput, low-latency serving engine for text-embeddings and various models","github_url":"https://github.com/michaelfeil/infinity","owner":"michaelfeil","repo":"infinity","owner_avatar_url":"https://avatars.githubusercontent.com/u/63565275?v=4","primary_language":"Python","stars":2907,"forks":196,"topics":["bert-embeddings","llm","text-embeddings"],"archived":false,"github_pushed_at":"2026-03-24T03:59:47+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/michaelfeil-infinity","markdown_url":"https://www.graphcanon.com/tools/michaelfeil-infinity.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/michaelfeil-infinity","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=michaelfeil-infinity"},"categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"clap","name":"clap"},{"slug":"clip","name":"clip"},{"slug":"colpali","name":"colpali"},{"slug":"docker-container","name":"docker-container"},{"slug":"gpu-acceleration","name":"gpu acceleration"},{"slug":"llm","name":"llm"},{"slug":"text-embeddings","name":"text-embeddings"}],"edges":[],"neighbours":[{"slug":"alexsjones-llmfit","name":"llmfit","tagline":"Hundreds of models & providers. One command to find what runs on your hardware.","github_url":"https://github.com/AlexsJones/llmfit","owner":"AlexsJones","repo":"llmfit","owner_avatar_url":"https://avatars.githubusercontent.com/u/1235925?v=4","primary_language":"Rust","stars":31867,"forks":1978,"topics":["gguf","llm","localai","mlx","skill","unsloth"],"archived":false,"github_pushed_at":"2026-08-14T07:36:41+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/alexsjones-llmfit","markdown_url":"https://www.graphcanon.com/tools/alexsjones-llmfit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alexsjones-llmfit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alexsjones-llmfit","shared_categories":[]},{"slug":"eugeneyan-open-llms","name":"open-llms","tagline":"A list of open LLMs available for commercial use.","github_url":"https://github.com/eugeneyan/open-llms","owner":"eugeneyan","repo":"open-llms","owner_avatar_url":"https://avatars.githubusercontent.com/u/6831355?v=4","primary_language":null,"stars":12849,"forks":985,"topics":["commercial","large-language-models","llm","llms"],"archived":false,"github_pushed_at":"2025-02-13T06:37:12+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/eugeneyan-open-llms","markdown_url":"https://www.graphcanon.com/tools/eugeneyan-open-llms.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eugeneyan-open-llms","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eugeneyan-open-llms","shared_categories":[]},{"slug":"jina-ai-clip-as-service","name":"clip-as-service","tagline":"-scalable embedding, reasoning, ranking for images and sentences with CLIP-","github_url":"https://github.com/jina-ai/clip-as-service","owner":"jina-ai","repo":"clip-as-service","owner_avatar_url":"https://avatars.githubusercontent.com/u/60539444?v=4","primary_language":"Python","stars":12834,"forks":2068,"topics":["bert","bert-as-service","clip-as-service","clip-model","cross-modal-retrieval","cross-modality","deep-learning","image2vec","multi-modality","neural-search","onnx","openai","pytorch","sentence-encoding","sentence2vec"],"archived":false,"github_pushed_at":"2024-01-23T10:33:43+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/jina-ai-clip-as-service","markdown_url":"https://www.graphcanon.com/tools/jina-ai-clip-as-service.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jina-ai-clip-as-service","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jina-ai-clip-as-service","shared_categories":[]},{"slug":"steven2358-awesome-generative-ai","name":"awesome-generative-ai","tagline":"A curated list of modern Generative Artificial Intelligence projects and services","github_url":"https://github.com/steven2358/awesome-generative-ai","owner":"steven2358","repo":"awesome-generative-ai","owner_avatar_url":"https://avatars.githubusercontent.com/u/164072?v=4","primary_language":null,"stars":12501,"forks":1990,"topics":["ai","artificial-intelligence","awesome","awesome-list","generative-ai","generative-art","large-language-models","llm"],"archived":false,"github_pushed_at":"2026-08-03T10:58:05+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/steven2358-awesome-generative-ai","markdown_url":"https://www.graphcanon.com/tools/steven2358-awesome-generative-ai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/steven2358-awesome-generative-ai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=steven2358-awesome-generative-ai","shared_categories":["inference-serving"]},{"slug":"andyyyy64-whichllm","name":"whichllm","tagline":"Command-line tool to find and benchmark local LLM performance","github_url":"https://github.com/Andyyyy64/whichllm","owner":"Andyyyy64","repo":"whichllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/105579829?v=4","primary_language":"Python","stars":6225,"forks":330,"topics":["ai","apple-silicon","benchmarks","cli","command-line-tool","gguf","gpu","huggingface","inference","llm","local-llm","ollama","python","vram"],"archived":false,"github_pushed_at":"2026-08-05T07:15:32+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/andyyyy64-whichllm","markdown_url":"https://www.graphcanon.com/tools/andyyyy64-whichllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/andyyyy64-whichllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=andyyyy64-whichllm","shared_categories":["inference-serving"]},{"slug":"huggingface-text-embeddings-inference","name":"text-embeddings-inference","tagline":"Blazing fast inference solution for text embeddings models","github_url":"https://github.com/huggingface/text-embeddings-inference","owner":"huggingface","repo":"text-embeddings-inference","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Rust","stars":4982,"forks":421,"topics":["ai","embeddings","huggingface","llm","ml"],"archived":false,"github_pushed_at":"2026-07-24T13:47:50+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huggingface-text-embeddings-inference","markdown_url":"https://www.graphcanon.com/tools/huggingface-text-embeddings-inference.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-text-embeddings-inference","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-text-embeddings-inference","shared_categories":["inference-serving"]},{"slug":"qdrant-fastembed","name":"fastembed","tagline":"Fast, Accurate, Lightweight Python library for creating state-of-the-art embeddings","github_url":"https://github.com/qdrant/fastembed","owner":"qdrant","repo":"fastembed","owner_avatar_url":"https://avatars.githubusercontent.com/u/73504361?v=4","primary_language":"Python","stars":3158,"forks":231,"topics":["embeddings","openai","rag","retrieval","retrieval-augmented-generation","vector-search"],"archived":false,"github_pushed_at":"2026-08-19T16:03:23+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/qdrant-fastembed","markdown_url":"https://www.graphcanon.com/tools/qdrant-fastembed.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/qdrant-fastembed","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=qdrant-fastembed","shared_categories":[]},{"slug":"b4rtaz-distributed-llama","name":"distributed-llama","tagline":"Distributed LLM inference using home devices cluster","github_url":"https://github.com/b4rtaz/distributed-llama","owner":"b4rtaz","repo":"distributed-llama","owner_avatar_url":"https://avatars.githubusercontent.com/u/12797776?v=4","primary_language":"C++","stars":3044,"forks":246,"topics":["distributed-computing","distributed-llm","llama2","llama3","llm","llm-inference","llms","neural-network","open-llm"],"archived":false,"github_pushed_at":"2026-07-05T16:47:20+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama","markdown_url":"https://www.graphcanon.com/tools/b4rtaz-distributed-llama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/b4rtaz-distributed-llama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=b4rtaz-distributed-llama","shared_categories":["inference-serving"]},{"slug":"superlinked-sie","name":"sie","tagline":"Open-source inference server and production cluster for all the models your agent needs.","github_url":"https://github.com/superlinked/sie","owner":"superlinked","repo":"sie","owner_avatar_url":"https://avatars.githubusercontent.com/u/94243920?v=4","primary_language":"Python","stars":2804,"forks":272,"topics":["bge","colbert","data-pipeline","deep-learning","embeddings","inference","inference-server","information-retrieval","llm","ml","mlops","natural-language-processing","nlp","python","reranking","retrieval","retrieval-augmented-generation","semantic-search","splade","vector-search"],"archived":false,"github_pushed_at":"2026-08-21T20:28:04+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/superlinked-sie","markdown_url":"https://www.graphcanon.com/tools/superlinked-sie.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/superlinked-sie","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=superlinked-sie","shared_categories":["inference-serving"]},{"slug":"huangowen-awesome-llm-compression","name":"Awesome-LLM-Compression","tagline":"Awesome LLM compression research papers and tools to accelerate LLM training and inference.","github_url":"https://github.com/HuangOwen/Awesome-LLM-Compression","owner":"HuangOwen","repo":"Awesome-LLM-Compression","owner_avatar_url":"https://avatars.githubusercontent.com/u/24937399?v=4","primary_language":null,"stars":1859,"forks":129,"topics":[],"archived":false,"github_pushed_at":"2026-06-30T15:26:46+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression","markdown_url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huangowen-awesome-llm-compression","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huangowen-awesome-llm-compression","shared_categories":["inference-serving"]},{"slug":"starlightsearch-embedanything","name":"EmbedAnything","tagline":"Highly Performant, Modular, Memory Safe and Production-ready Inference, Ingestion and Indexing built in Rust","github_url":"https://github.com/StarlightSearch/EmbedAnything","owner":"StarlightSearch","repo":"EmbedAnything","owner_avatar_url":"https://avatars.githubusercontent.com/u/165606246?v=4","primary_language":"Rust","stars":1304,"forks":143,"topics":["ai","cloud","generative-ai","hacktoberfest","high-performance","indexing","inference","information-retrieval","large-language-models","local","machine-learning","onnxruntime","pipeline","production-ready","python","rag","rust","search","server","vector-database"],"archived":false,"github_pushed_at":"2026-08-12T08:56:59+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/starlightsearch-embedanything","markdown_url":"https://www.graphcanon.com/tools/starlightsearch-embedanything.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/starlightsearch-embedanything","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=starlightsearch-embedanything","shared_categories":["inference-serving"]},{"slug":"kubeai-project-kubeai","name":"kubeai","tagline":"AI Inference Operator for Kubernetes","github_url":"https://github.com/kubeai-project/kubeai","owner":"kubeai-project","repo":"kubeai","owner_avatar_url":"https://avatars.githubusercontent.com/u/232319222?v=4","primary_language":"Go","stars":1237,"forks":131,"topics":["ai","autoscaler","faster-whisper","inference-operator","k8s","kubernetes","llm","ollama","ollama-operator","openai-api","vllm","vllm-operator","whisper"],"archived":false,"github_pushed_at":"2026-07-31T01:04:47+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/kubeai-project-kubeai","markdown_url":"https://www.graphcanon.com/tools/kubeai-project-kubeai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kubeai-project-kubeai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kubeai-project-kubeai","shared_categories":["inference-serving"]}]}}