{"data":{"node":{"slug":"langchain-ai-auto-evaluator","name":"auto-evaluator","tagline":"auto-evaluator","github_url":"https://github.com/langchain-ai/auto-evaluator","owner":"langchain-ai","repo":"auto-evaluator","owner_avatar_url":"https://avatars.githubusercontent.com/u/126733545?v=4","primary_language":"TypeScript","stars":783,"forks":102,"topics":[],"archived":true,"github_pushed_at":"2025-06-26T03:58:55+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/langchain-ai-auto-evaluator","markdown_url":"https://www.graphcanon.com/tools/langchain-ai-auto-evaluator.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langchain-ai-auto-evaluator","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langchain-ai-auto-evaluator"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"auto-evaluation","name":"auto-evaluation"},{"slug":"railway","name":"railway"},{"slug":"typescript","name":"typescript"},{"slug":"vercel","name":"vercel"}],"edges":[],"neighbours":[{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":17226,"forks":1736,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-07-27T11:33:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"giskard-ai-giskard-oss","name":"giskard-oss","tagline":"Open-Source Evaluation & Testing library for LLM Agents","github_url":"https://github.com/Giskard-AI/giskard-oss","owner":"Giskard-AI","repo":"giskard-oss","owner_avatar_url":"https://avatars.githubusercontent.com/u/71782571?v=4","primary_language":"Python","stars":5727,"forks":511,"topics":["agent-evaluation","ai-red-team","ai-security","ai-testing","fairness-ai","llm","llm-eval","llm-evaluation","llm-security","llmops","ml-testing","ml-validation","mlops","rag-evaluation","red-team-tools","responsible-ai","trustworthy-ai"],"archived":false,"github_pushed_at":"2026-08-01T23:22:37+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/giskard-ai-giskard-oss","markdown_url":"https://www.graphcanon.com/tools/giskard-ai-giskard-oss.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/giskard-ai-giskard-oss","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=giskard-ai-giskard-oss","shared_categories":["evaluation-observability"]},{"slug":"openai-simple-evals","name":"simple-evals","tagline":"A lightweight library for evaluating language models.","github_url":"https://github.com/openai/simple-evals","owner":"openai","repo":"simple-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/14957082?v=4","primary_language":"Python","stars":4595,"forks":501,"topics":[],"archived":false,"github_pushed_at":"2026-04-22T22:16:18+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/openai-simple-evals","markdown_url":"https://www.graphcanon.com/tools/openai-simple-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openai-simple-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openai-simple-evals","shared_categories":["evaluation-observability"]},{"slug":"evalplus-evalplus","name":"evalplus","tagline":"Rigorous evaluation of LLM-synthesized code","github_url":"https://github.com/evalplus/evalplus","owner":"evalplus","repo":"evalplus","owner_avatar_url":"https://avatars.githubusercontent.com/u/132106461?v=4","primary_language":"Python","stars":1794,"forks":205,"topics":["benchmark","chatgpt","efficiency","gpt-4","large-language-models","program-synthesis","testing"],"archived":false,"github_pushed_at":"2025-10-02T22:56:38+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/evalplus-evalplus","markdown_url":"https://www.graphcanon.com/tools/evalplus-evalplus.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evalplus-evalplus","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evalplus-evalplus","shared_categories":["evaluation-observability"]},{"slug":"rlancemartin-auto-evaluator","name":"auto-evaluator","tagline":"A lightweight evaluation tool for question-answering using Langchain","github_url":"https://github.com/rlancemartin/auto-evaluator","owner":"rlancemartin","repo":"auto-evaluator","owner_avatar_url":"https://avatars.githubusercontent.com/u/122662504?v=4","primary_language":"Python","stars":1105,"forks":92,"topics":[],"archived":false,"github_pushed_at":"2023-05-10T02:00:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator","markdown_url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rlancemartin-auto-evaluator","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rlancemartin-auto-evaluator","shared_categories":["evaluation-observability"]},{"slug":"bigcode-project-bigcode-evaluation-harness","name":"bigcode-evaluation-harness","tagline":"A framework for evaluating autoregressive code generation language models.","github_url":"https://github.com/bigcode-project/bigcode-evaluation-harness","owner":"bigcode-project","repo":"bigcode-evaluation-harness","owner_avatar_url":"https://avatars.githubusercontent.com/u/110470554?v=4","primary_language":"Python","stars":1055,"forks":261,"topics":[],"archived":false,"github_pushed_at":"2025-07-22T13:18:09+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/bigcode-project-bigcode-evaluation-harness","markdown_url":"https://www.graphcanon.com/tools/bigcode-project-bigcode-evaluation-harness.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bigcode-project-bigcode-evaluation-harness","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bigcode-project-bigcode-evaluation-harness","shared_categories":["evaluation-observability"]},{"slug":"benchflow-ai-awesome-evals","name":"awesome-evals","tagline":"A curated library of resources for building and evaluating AI agents","github_url":"https://github.com/benchflow-ai/awesome-evals","owner":"benchflow-ai","repo":"awesome-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/190338344?v=4","primary_language":null,"stars":761,"forks":71,"topics":["agent-evaluation","ai-agents","awesome","awesome-list","benchmarks","evals","llm","llm-evaluation","rl-environments"],"archived":false,"github_pushed_at":"2026-07-01T22:53:19+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals","markdown_url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/benchflow-ai-awesome-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=benchflow-ai-awesome-evals","shared_categories":["evaluation-observability"]},{"slug":"declare-lab-instruct-eval","name":"instruct-eval","tagline":"Quantitative evaluation for instruction-tuned language models","github_url":"https://github.com/declare-lab/instruct-eval","owner":"declare-lab","repo":"instruct-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/59164695?v=4","primary_language":"Python","stars":552,"forks":45,"topics":["instruct-tuning","llm"],"archived":false,"github_pushed_at":"2024-03-10T05:00:00+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval","markdown_url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/declare-lab-instruct-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=declare-lab-instruct-eval","shared_categories":["evaluation-observability"]},{"slug":"verifywise-ai-verifywise","name":"verifywise","tagline":"Complete AI governance and LLM Evals platform","github_url":"https://github.com/verifywise-ai/verifywise","owner":"verifywise-ai","repo":"verifywise","owner_avatar_url":"https://avatars.githubusercontent.com/u/262239808?v=4","primary_language":"TypeScript","stars":322,"forks":110,"topics":["ai","ai-auditing","ai-compliance","ai-governance","ai-governance-model","ai-risk","audit","auditing","compliance","eu-ai-act","governance","grc","iso27001","iso42001","llm-eval","llm-evaluation","nist-ai-rmf","risk-management"],"archived":false,"github_pushed_at":"2026-07-28T09:05:06+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/verifywise-ai-verifywise","markdown_url":"https://www.graphcanon.com/tools/verifywise-ai-verifywise.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/verifywise-ai-verifywise","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=verifywise-ai-verifywise","shared_categories":["evaluation-observability"]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]},{"slug":"zeno-ml-zeno","name":"zeno","tagline":"AI Data Management & Evaluation Platform","github_url":"https://github.com/zeno-ml/zeno","owner":"zeno-ml","repo":"zeno","owner_avatar_url":"https://avatars.githubusercontent.com/u/109821189?v=4","primary_language":"Svelte","stars":214,"forks":11,"topics":["ai","data-science","evaluation","evaluation-framework","machine-learning","python"],"archived":true,"github_pushed_at":"2023-10-05T19:02:16+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/zeno-ml-zeno","markdown_url":"https://www.graphcanon.com/tools/zeno-ml-zeno.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zeno-ml-zeno","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zeno-ml-zeno","shared_categories":["evaluation-observability"]},{"slug":"hidai25-eval-view","name":"eval-view","tagline":"Regression testing for AI agents","github_url":"https://github.com/hidai25/eval-view","owner":"hidai25","repo":"eval-view","owner_avatar_url":"https://avatars.githubusercontent.com/u/31502796?v=4","primary_language":"Python","stars":126,"forks":21,"topics":["agent-benchmark","agent-evaluation","agentic-ai","ai-agents","anthropic","autogen","cli","crewai","evaluation","langchain-agent","langgraph","llm","mcp","openai-assistants","pytest","python","regression-testing","testing"],"archived":false,"github_pushed_at":"2026-07-26T19:53:19+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/hidai25-eval-view","markdown_url":"https://www.graphcanon.com/tools/hidai25-eval-view.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/hidai25-eval-view","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=hidai25-eval-view","shared_categories":["evaluation-observability"]}]}}