{"data":{"node":{"slug":"evaleval-every-eval-ever","name":"every_eval_ever","tagline":"Shared schema and crowdsourced eval database","github_url":"https://github.com/evaleval/every_eval_ever","owner":"evaleval","repo":"every_eval_ever","owner_avatar_url":"https://avatars.githubusercontent.com/u/176316740?v=4","primary_language":"Python","stars":111,"forks":49,"topics":["agent-evaluation","ai-evaluation","evaluations","infra","llm-evaluation"],"archived":false,"github_pushed_at":"2026-09-07T11:57:22+00:00","maintenance_label":"Very active","stars_delta_30d":9,"url":"https://www.graphcanon.com/tools/evaleval-every-eval-ever","markdown_url":"https://www.graphcanon.com/tools/evaleval-every-eval-ever.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evaleval-every-eval-ever","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evaleval-every-eval-ever"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"agent-evaluation","name":"agent-evaluation"},{"slug":"ai-evaluation","name":"ai-evaluation"},{"slug":"evaluations","name":"evaluations"},{"slug":"infra","name":"infra"},{"slug":"llm-evaluation","name":"llm-evaluation"}],"edges":[],"neighbours":[{"slug":"openai-evals","name":"evals","tagline":"Framework for evaluating LLMs and LLM systems with an open-source registry of benchmarks.","github_url":"https://github.com/openai/evals","owner":"openai","repo":"evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/14957082?v=4","primary_language":"Python","stars":19398,"forks":3077,"topics":[],"archived":false,"github_pushed_at":"2026-04-14T15:29:57+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/openai-evals","markdown_url":"https://www.graphcanon.com/tools/openai-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openai-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openai-evals","shared_categories":["evaluation-observability"]},{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":18342,"forks":1953,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-09-18T17:06:58+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"openai-simple-evals","name":"simple-evals","tagline":"A lightweight library for evaluating language models.","github_url":"https://github.com/openai/simple-evals","owner":"openai","repo":"simple-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/14957082?v=4","primary_language":"Python","stars":4625,"forks":508,"topics":[],"archived":false,"github_pushed_at":"2026-04-22T22:16:18+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/openai-simple-evals","markdown_url":"https://www.graphcanon.com/tools/openai-simple-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openai-simple-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openai-simple-evals","shared_categories":["evaluation-observability"]},{"slug":"stanford-crfm-helm","name":"helm","tagline":"Holistic, reproducible and transparent evaluation of foundation models","github_url":"https://github.com/stanford-crfm/helm","owner":"stanford-crfm","repo":"helm","owner_avatar_url":"https://avatars.githubusercontent.com/u/75054807?v=4","primary_language":"Python","stars":2902,"forks":411,"topics":[],"archived":false,"github_pushed_at":"2026-09-01T01:33:19+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/stanford-crfm-helm","markdown_url":"https://www.graphcanon.com/tools/stanford-crfm-helm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/stanford-crfm-helm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=stanford-crfm-helm","shared_categories":["evaluation-observability"]},{"slug":"huggingface-lighteval","name":"lighteval","tagline":"All-in-one toolkit for evaluating LLMs across multiple backends","github_url":"https://github.com/huggingface/lighteval","owner":"huggingface","repo":"lighteval","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":2535,"forks":552,"topics":["evaluation","evaluation-framework","evaluation-metrics","huggingface"],"archived":false,"github_pushed_at":"2026-08-11T13:10:37+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huggingface-lighteval","markdown_url":"https://www.graphcanon.com/tools/huggingface-lighteval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-lighteval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-lighteval","shared_categories":["evaluation-observability"]},{"slug":"evalplus-evalplus","name":"evalplus","tagline":"Rigorous evaluation of LLM-synthesized code","github_url":"https://github.com/evalplus/evalplus","owner":"evalplus","repo":"evalplus","owner_avatar_url":"https://avatars.githubusercontent.com/u/132106461?v=4","primary_language":"Python","stars":1806,"forks":206,"topics":["benchmark","chatgpt","efficiency","gpt-4","large-language-models","program-synthesis","testing"],"archived":false,"github_pushed_at":"2025-10-02T22:56:38+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/evalplus-evalplus","markdown_url":"https://www.graphcanon.com/tools/evalplus-evalplus.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evalplus-evalplus","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evalplus-evalplus","shared_categories":["evaluation-observability"]},{"slug":"rlancemartin-auto-evaluator","name":"auto-evaluator","tagline":"A lightweight evaluation tool for question-answering using Langchain","github_url":"https://github.com/rlancemartin/auto-evaluator","owner":"rlancemartin","repo":"auto-evaluator","owner_avatar_url":"https://avatars.githubusercontent.com/u/122662504?v=4","primary_language":"Python","stars":1102,"forks":92,"topics":[],"archived":false,"github_pushed_at":"2023-05-10T02:00:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator","markdown_url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rlancemartin-auto-evaluator","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rlancemartin-auto-evaluator","shared_categories":["evaluation-observability"]},{"slug":"benchflow-ai-awesome-evals","name":"awesome-evals","tagline":"A curated library of resources for building and evaluating AI agents","github_url":"https://github.com/benchflow-ai/awesome-evals","owner":"benchflow-ai","repo":"awesome-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/190338344?v=4","primary_language":null,"stars":901,"forks":104,"topics":["agent-evaluation","ai-agents","awesome","awesome-list","benchmarks","evals","llm","llm-evaluation","rl-environments"],"archived":false,"github_pushed_at":"2026-09-15T17:49:19+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals","markdown_url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/benchflow-ai-awesome-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=benchflow-ai-awesome-evals","shared_categories":["evaluation-observability"]},{"slug":"langchain-ai-auto-evaluator","name":"auto-evaluator","tagline":"auto-evaluator","github_url":"https://github.com/langchain-ai/auto-evaluator","owner":"langchain-ai","repo":"auto-evaluator","owner_avatar_url":"https://avatars.githubusercontent.com/u/126733545?v=4","primary_language":"TypeScript","stars":783,"forks":99,"topics":[],"archived":true,"github_pushed_at":"2025-06-26T03:58:55+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/langchain-ai-auto-evaluator","markdown_url":"https://www.graphcanon.com/tools/langchain-ai-auto-evaluator.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langchain-ai-auto-evaluator","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langchain-ai-auto-evaluator","shared_categories":["evaluation-observability"]},{"slug":"verifywise-ai-verifywise","name":"verifywise","tagline":"Complete AI governance and LLM Evals platform","github_url":"https://github.com/verifywise-ai/verifywise","owner":"verifywise-ai","repo":"verifywise","owner_avatar_url":"https://avatars.githubusercontent.com/u/262239808?v=4","primary_language":"TypeScript","stars":354,"forks":117,"topics":["ai","ai-auditing","ai-compliance","ai-governance","ai-governance-model","ai-risk","audit","auditing","compliance","eu-ai-act","governance","grc","iso27001","iso42001","llm-eval","llm-evaluation","nist-ai-rmf","risk-management"],"archived":false,"github_pushed_at":"2026-09-19T19:03:23+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/verifywise-ai-verifywise","markdown_url":"https://www.graphcanon.com/tools/verifywise-ai-verifywise.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/verifywise-ai-verifywise","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=verifywise-ai-verifywise","shared_categories":["evaluation-observability"]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]},{"slug":"jinjieni-mixeval","name":"MixEval","tagline":"Evaluation suite and dynamic data release for MixEval","github_url":"https://github.com/JinjieNi/MixEval","owner":"JinjieNi","repo":"MixEval","owner_avatar_url":"https://avatars.githubusercontent.com/u/46987361?v=4","primary_language":"Python","stars":254,"forks":40,"topics":["benchmark","benchmark-mixture","benchmarking-framework","benchmarking-suite","evaluation","evaluation-framework","foundation-models","large-language-model","large-language-models","large-multimodal-models","llm-evaluation","llm-evaluation-framework","llm-inference","mixeval"],"archived":false,"github_pushed_at":"2024-11-10T02:23:50+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/jinjieni-mixeval","markdown_url":"https://www.graphcanon.com/tools/jinjieni-mixeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jinjieni-mixeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jinjieni-mixeval","shared_categories":["evaluation-observability"]}]}}