{"data":{"node":{"slug":"huggingface-lighteval","name":"lighteval","tagline":"All-in-one toolkit for evaluating LLMs across multiple backends","github_url":"https://github.com/huggingface/lighteval","owner":"huggingface","repo":"lighteval","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":2508,"forks":523,"topics":["evaluation","evaluation-framework","evaluation-metrics","huggingface"],"archived":false,"github_pushed_at":"2026-06-29T13:03:33+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huggingface-lighteval","markdown_url":"https://www.graphcanon.com/tools/huggingface-lighteval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-lighteval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-lighteval"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"evaluation","name":"evaluation"},{"slug":"evaluation-framework","name":"evaluation-framework"},{"slug":"evaluation-metrics","name":"evaluation-metrics"},{"slug":"huggingface","name":"huggingface"},{"slug":"python","name":"python"}],"edges":[],"neighbours":[{"slug":"langfuse-langfuse","name":"langfuse","tagline":"Open source AI engineering platform: LLM evals, observability, metrics, prompt management, playground, datasets","github_url":"https://github.com/langfuse/langfuse","owner":"langfuse","repo":"langfuse","owner_avatar_url":"https://avatars.githubusercontent.com/u/134601687?v=4","primary_language":"TypeScript","stars":32271,"forks":3466,"topics":["analytics","autogen","evaluation","langchain","large-language-models","llama-index","llm","llm-evaluation","llm-observability","llmops","monitoring","observability","open-source","openai","playground","prompt-engineering","prompt-management","self-hosted","ycombinator"],"archived":false,"github_pushed_at":"2026-07-31T22:58:07+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/langfuse-langfuse","markdown_url":"https://www.graphcanon.com/tools/langfuse-langfuse.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langfuse-langfuse","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langfuse-langfuse","shared_categories":["evaluation-observability"]},{"slug":"openai-evals","name":"evals","tagline":"Framework for evaluating LLMs and LLM systems with an open-source registry of benchmarks.","github_url":"https://github.com/openai/evals","owner":"openai","repo":"evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/14957082?v=4","primary_language":"Python","stars":19127,"forks":3050,"topics":[],"archived":false,"github_pushed_at":"2026-04-14T15:29:57+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/openai-evals","markdown_url":"https://www.graphcanon.com/tools/openai-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openai-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openai-evals","shared_categories":["evaluation-observability"]},{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":17226,"forks":1736,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-07-27T11:33:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"lightning-ai-litgpt","name":"litgpt","tagline":"High-performance LLMs with recipes for pretraining, finetuning and deployment","github_url":"https://github.com/Lightning-AI/litgpt","owner":"Lightning-AI","repo":"litgpt","owner_avatar_url":"https://avatars.githubusercontent.com/u/58386951?v=4","primary_language":"Python","stars":13605,"forks":1483,"topics":["ai","artificial-intelligence","deep-learning","large-language-models","llm","llm-inference","llms"],"archived":false,"github_pushed_at":"2026-07-20T10:24:12+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lightning-ai-litgpt","markdown_url":"https://www.graphcanon.com/tools/lightning-ai-litgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lightning-ai-litgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lightning-ai-litgpt","shared_categories":[]},{"slug":"eleutherai-lm-evaluation-harness","name":"lm-evaluation-harness","tagline":"A framework for few-shot evaluation of language models.","github_url":"https://github.com/EleutherAI/lm-evaluation-harness","owner":"EleutherAI","repo":"lm-evaluation-harness","owner_avatar_url":"https://avatars.githubusercontent.com/u/68924597?v=4","primary_language":"Python","stars":13560,"forks":3467,"topics":["evaluation-framework","language-model","transformer"],"archived":false,"github_pushed_at":"2026-07-13T20:18:15+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness","markdown_url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eleutherai-lm-evaluation-harness","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eleutherai-lm-evaluation-harness","shared_categories":["evaluation-observability"]},{"slug":"evolvinglmms-lab-lmms-eval","name":"lmms-eval","tagline":"One-for-All Multimodal Evaluation Toolkit Across Text, Image, Video, and Audio Tasks","github_url":"https://github.com/EvolvingLMMs-Lab/lmms-eval","owner":"EvolvingLMMs-Lab","repo":"lmms-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/154951679?v=4","primary_language":"Python","stars":4368,"forks":639,"topics":["agi","audio-evaluation","benchmark","evaluation","large-language-models","llm-evaluation","multimodal","multimodal-evaluation","video-understanding","vision-language-model","vlm"],"archived":false,"github_pushed_at":"2026-08-06T02:22:23+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/evolvinglmms-lab-lmms-eval","markdown_url":"https://www.graphcanon.com/tools/evolvinglmms-lab-lmms-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evolvinglmms-lab-lmms-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evolvinglmms-lab-lmms-eval","shared_categories":["evaluation-observability"]},{"slug":"stanford-crfm-helm","name":"helm","tagline":"Holistic, reproducible and transparent evaluation of foundation models","github_url":"https://github.com/stanford-crfm/helm","owner":"stanford-crfm","repo":"helm","owner_avatar_url":"https://avatars.githubusercontent.com/u/75054807?v=4","primary_language":"Python","stars":2873,"forks":406,"topics":[],"archived":false,"github_pushed_at":"2026-08-01T01:23:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/stanford-crfm-helm","markdown_url":"https://www.graphcanon.com/tools/stanford-crfm-helm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/stanford-crfm-helm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=stanford-crfm-helm","shared_categories":["evaluation-observability"]},{"slug":"evalplus-evalplus","name":"evalplus","tagline":"Rigorous evaluation of LLM-synthesized code","github_url":"https://github.com/evalplus/evalplus","owner":"evalplus","repo":"evalplus","owner_avatar_url":"https://avatars.githubusercontent.com/u/132106461?v=4","primary_language":"Python","stars":1794,"forks":205,"topics":["benchmark","chatgpt","efficiency","gpt-4","large-language-models","program-synthesis","testing"],"archived":false,"github_pushed_at":"2025-10-02T22:56:38+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/evalplus-evalplus","markdown_url":"https://www.graphcanon.com/tools/evalplus-evalplus.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evalplus-evalplus","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evalplus-evalplus","shared_categories":["evaluation-observability"]},{"slug":"rlancemartin-auto-evaluator","name":"auto-evaluator","tagline":"A lightweight evaluation tool for question-answering using Langchain","github_url":"https://github.com/rlancemartin/auto-evaluator","owner":"rlancemartin","repo":"auto-evaluator","owner_avatar_url":"https://avatars.githubusercontent.com/u/122662504?v=4","primary_language":"Python","stars":1105,"forks":92,"topics":[],"archived":false,"github_pushed_at":"2023-05-10T02:00:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator","markdown_url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rlancemartin-auto-evaluator","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rlancemartin-auto-evaluator","shared_categories":["evaluation-observability"]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]},{"slug":"alopatenko-llmevaluation","name":"LLMEvaluation","tagline":"A comprehensive guide to LLM evaluation methods","github_url":"https://github.com/alopatenko/LLMEvaluation","owner":"alopatenko","repo":"LLMEvaluation","owner_avatar_url":"https://avatars.githubusercontent.com/u/7122933?v=4","primary_language":"HTML","stars":196,"forks":22,"topics":["evaluation","generative-ai-benchmarking","llm","llm-benchmarking","llm-evaluation"],"archived":false,"github_pushed_at":"2026-07-06T01:17:36+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation","markdown_url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alopatenko-llmevaluation","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alopatenko-llmevaluation","shared_categories":["evaluation-observability"]},{"slug":"raga-ai-hub-raga-llm-hub","name":"raga-llm-hub","tagline":"Framework for LLM evaluation, guardrails and security","github_url":"https://github.com/raga-ai-hub/raga-llm-hub","owner":"raga-ai-hub","repo":"raga-llm-hub","owner_avatar_url":"https://avatars.githubusercontent.com/u/161833182?v=4","primary_language":"Python","stars":114,"forks":14,"topics":["guardrails","llm-evaluation","llm-security","llmops"],"archived":false,"github_pushed_at":"2024-09-09T10:53:31+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/raga-ai-hub-raga-llm-hub","markdown_url":"https://www.graphcanon.com/tools/raga-ai-hub-raga-llm-hub.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/raga-ai-hub-raga-llm-hub","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=raga-ai-hub-raga-llm-hub","shared_categories":["evaluation-observability"]}]}}