{"data":{"node":{"slug":"alphadl-adarubrics","name":"AdaRubrics","tagline":"Adaptive Dynamic Rubric Evaluator for Agent Trajectories","github_url":"https://github.com/alphadl/AdaRubrics","owner":"alphadl","repo":"AdaRubrics","owner_avatar_url":"https://avatars.githubusercontent.com/u/20458732?v=4","primary_language":"Python","stars":345,"forks":36,"topics":["agent-evaluation","llm-evaluation","reward-model","rlhf","rubric"],"archived":false,"github_pushed_at":"2026-06-07T12:28:48+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/alphadl-adarubrics","markdown_url":"https://www.graphcanon.com/tools/alphadl-adarubrics.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alphadl-adarubrics","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alphadl-adarubrics"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"agent-evaluation","name":"agent-evaluation"},{"slug":"llm-evaluation","name":"llm-evaluation"},{"slug":"reward-model","name":"reward-model"},{"slug":"rlhf","name":"rlhf"},{"slug":"rubric","name":"rubric"}],"edges":[],"neighbours":[{"slug":"crewaiinc-crewai","name":"crewAI","tagline":"Framework for orchestrating role-playing AI agents","github_url":"https://github.com/crewAIInc/crewAI","owner":"crewAIInc","repo":"crewAI","owner_avatar_url":"https://avatars.githubusercontent.com/u/170677839?v=4","primary_language":"Python","stars":56779,"forks":8094,"topics":["agents","ai","ai-agents","aiagentframework","llms"],"archived":false,"github_pushed_at":"2026-08-08T07:26:39+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/crewaiinc-crewai","markdown_url":"https://www.graphcanon.com/tools/crewaiinc-crewai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/crewaiinc-crewai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=crewaiinc-crewai","shared_categories":[]},{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":17226,"forks":1736,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-07-27T11:33:31+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"openbmb-xagent","name":"XAgent","tagline":"An Autonomous LLM Agent for Complex Task Solving","github_url":"https://github.com/OpenBMB/XAgent","owner":"OpenBMB","repo":"XAgent","owner_avatar_url":"https://avatars.githubusercontent.com/u/89920203?v=4","primary_language":"Python","stars":8534,"forks":904,"topics":[],"archived":false,"github_pushed_at":"2026-07-31T03:30:54+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/openbmb-xagent","markdown_url":"https://www.graphcanon.com/tools/openbmb-xagent.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openbmb-xagent","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openbmb-xagent","shared_categories":[]},{"slug":"giskard-ai-giskard-oss","name":"giskard-oss","tagline":"Open-Source Evaluation & Testing library for LLM Agents","github_url":"https://github.com/Giskard-AI/giskard-oss","owner":"Giskard-AI","repo":"giskard-oss","owner_avatar_url":"https://avatars.githubusercontent.com/u/71782571?v=4","primary_language":"Python","stars":5727,"forks":511,"topics":["agent-evaluation","ai-red-team","ai-security","ai-testing","fairness-ai","llm","llm-eval","llm-evaluation","llm-security","llmops","ml-testing","ml-validation","mlops","rag-evaluation","red-team-tools","responsible-ai","trustworthy-ai"],"archived":false,"github_pushed_at":"2026-08-01T23:22:37+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/giskard-ai-giskard-oss","markdown_url":"https://www.graphcanon.com/tools/giskard-ai-giskard-oss.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/giskard-ai-giskard-oss","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=giskard-ai-giskard-oss","shared_categories":["evaluation-observability"]},{"slug":"evalplus-evalplus","name":"evalplus","tagline":"Rigorous evaluation of LLM-synthesized code","github_url":"https://github.com/evalplus/evalplus","owner":"evalplus","repo":"evalplus","owner_avatar_url":"https://avatars.githubusercontent.com/u/132106461?v=4","primary_language":"Python","stars":1794,"forks":205,"topics":["benchmark","chatgpt","efficiency","gpt-4","large-language-models","program-synthesis","testing"],"archived":false,"github_pushed_at":"2025-10-02T22:56:38+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/evalplus-evalplus","markdown_url":"https://www.graphcanon.com/tools/evalplus-evalplus.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evalplus-evalplus","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evalplus-evalplus","shared_categories":["evaluation-observability"]},{"slug":"future-agi-future-agi","name":"future-agi","tagline":"End-to-end platform for evaluating, observing, and improving LLM and AI agent applications","github_url":"https://github.com/future-agi/future-agi","owner":"future-agi","repo":"future-agi","owner_avatar_url":"https://avatars.githubusercontent.com/u/147392366?v=4","primary_language":"Python","stars":1559,"forks":449,"topics":["ai-agents","ai-evals","ai-gateway","ai-optimization","ai-simulations","evaluation-framework","guardrails","hallucination-detection","llm","llm-evaluation","llm-observability","llmops","model-evaluation","observability","opentelemetry","rag","rag-evaluation","simulation","telemetry","tracing"],"archived":false,"github_pushed_at":"2026-08-01T14:44:07+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/future-agi-future-agi","markdown_url":"https://www.graphcanon.com/tools/future-agi-future-agi.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/future-agi-future-agi","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=future-agi-future-agi","shared_categories":["evaluation-observability"]},{"slug":"benchflow-ai-awesome-evals","name":"awesome-evals","tagline":"A curated library of resources for building and evaluating AI agents","github_url":"https://github.com/benchflow-ai/awesome-evals","owner":"benchflow-ai","repo":"awesome-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/190338344?v=4","primary_language":null,"stars":761,"forks":71,"topics":["agent-evaluation","ai-agents","awesome","awesome-list","benchmarks","evals","llm","llm-evaluation","rl-environments"],"archived":false,"github_pushed_at":"2026-07-01T22:53:19+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals","markdown_url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/benchflow-ai-awesome-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=benchflow-ai-awesome-evals","shared_categories":["evaluation-observability"]},{"slug":"ethz-spylab-agentdojo","name":"agentdojo","tagline":"A Dynamic Environment to Evaluate Prompt Injection Attacks and Defenses for LLM Agents","github_url":"https://github.com/ethz-spylab/agentdojo","owner":"ethz-spylab","repo":"agentdojo","owner_avatar_url":"https://avatars.githubusercontent.com/u/106388551?v=4","primary_language":"Python","stars":716,"forks":188,"topics":["benchmark","large-language-models","prompt-injection","security"],"archived":false,"github_pushed_at":"2026-06-02T10:01:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/ethz-spylab-agentdojo","markdown_url":"https://www.graphcanon.com/tools/ethz-spylab-agentdojo.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ethz-spylab-agentdojo","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ethz-spylab-agentdojo","shared_categories":["evaluation-observability"]},{"slug":"oxbshw-llm-agents-ecosystem-handbook","name":"LLM-Agents-Ecosystem-Handbook","tagline":"One-stop handbook for building, deploying, and understanding LLM agents","github_url":"https://github.com/oxbshw/LLM-Agents-Ecosystem-Handbook","owner":"oxbshw","repo":"LLM-Agents-Ecosystem-Handbook","owner_avatar_url":"https://avatars.githubusercontent.com/u/212214682?v=4","primary_language":"Python","stars":539,"forks":85,"topics":["ai","ai-agent","ai-agents","fine-tuning","finetuning-llms","freamework","llm","llmops","local-development","mcp-server","memory","rag","rag-chatbot","voice-agent"],"archived":false,"github_pushed_at":"2026-06-30T12:22:57+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/oxbshw-llm-agents-ecosystem-handbook","markdown_url":"https://www.graphcanon.com/tools/oxbshw-llm-agents-ecosystem-handbook.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/oxbshw-llm-agents-ecosystem-handbook","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=oxbshw-llm-agents-ecosystem-handbook","shared_categories":["evaluation-observability"]},{"slug":"zju-vipa-odyssey","name":"Odyssey","tagline":"Empowering Minecraft Agents with Open-World Skills","github_url":"https://github.com/zju-vipa/Odyssey","owner":"zju-vipa","repo":"Odyssey","owner_avatar_url":"https://avatars.githubusercontent.com/u/43666493?v=4","primary_language":"Python","stars":402,"forks":29,"topics":["agent","embodied-agent","fine-tuning","large-language-model","large-language-models","llm","llm-agent","minecraft"],"archived":false,"github_pushed_at":"2025-10-22T07:37:45+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/zju-vipa-odyssey","markdown_url":"https://www.graphcanon.com/tools/zju-vipa-odyssey.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zju-vipa-odyssey","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zju-vipa-odyssey","shared_categories":[]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]},{"slug":"leoyeai-myclaw-bench","name":"myclaw-bench","tagline":"Benchmark for AI agents on OpenClaw","github_url":"https://github.com/LeoYeAI/myclaw-bench","owner":"LeoYeAI","repo":"myclaw-bench","owner_avatar_url":"https://avatars.githubusercontent.com/u/121472001?v=4","primary_language":"Python","stars":227,"forks":38,"topics":["agent-testing","ai-agent","ai-benchmark","benchmark","llm-evaluation","myclaw","openclaw"],"archived":false,"github_pushed_at":"2026-07-20T08:48:41+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/leoyeai-myclaw-bench","markdown_url":"https://www.graphcanon.com/tools/leoyeai-myclaw-bench.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/leoyeai-myclaw-bench","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=leoyeai-myclaw-bench","shared_categories":["evaluation-observability"]}]}}