{"data":{"slug":"benchflow-ai-awesome-evals","name":"awesome-evals","tagline":"A curated library of resources for building and evaluating AI agents","github_url":"https://github.com/benchflow-ai/awesome-evals","owner":"benchflow-ai","repo":"awesome-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/190338344?v=4","primary_language":null,"stars":761,"forks":71,"topics":["agent-evaluation","ai-agents","awesome","awesome-list","benchmarks","evals","llm","llm-evaluation","rl-environments"],"archived":false,"github_pushed_at":"2026-07-01T22:53:19+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals","markdown_url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/benchflow-ai-awesome-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=benchflow-ai-awesome-evals","description":"A curated, non-BS library of the best resources for building and evaluating AI agents — papers, blogs, talks, tools, benchmarks. Maintained by BenchFlow.","homepage_url":null,"license":"Other","open_issues":21,"watchers":2,"ai_summary":"Maintained by BenchFlow, this resource offers papers, blogs, talks, tools, and benchmarks focused on agent evaluation, featuring components related to AI agents, LLMs, and RL environments.","readme_excerpt":"## 5 · Evaluation infrastructure (the eval stack: datasets, scorers, online/offline, tracing, CI)\n\n*(All repos URL-verified via GitHub API, Jun 2026. `🆕` = released/expanded 2025–2026. `⚠️` = caveat/discontinued.)*\n\n---\n\n## License\n\n\n\nTo the extent possible under law, [BenchFlow](https://benchflow.ai) and contributors have waived all copyright and related rights to this work (CC0 1.0). The linked resources remain under their respective licenses.","github_created_at":"2026-06-24T08:10:33+00:00","created_at":"2026-07-11T11:59:46.573311+00:00","updated_at":"2026-07-28T12:00:50.101344+00:00","categories":[{"slug":"ai-agents","name":"AI Agents","url":"https://www.graphcanon.com/categories/ai-agents","markdown_url":"https://www.graphcanon.com/categories/ai-agents.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/ai-agents"},{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"agent-evaluation","name":"agent-evaluation"},{"slug":"ai-agents","name":"ai-agents"},{"slug":"awesome-list","name":"awesome-list"},{"slug":"benchmarks","name":"benchmarks"},{"slug":"llm-evaluation","name":"llm-evaluation"},{"slug":"rl-environments","name":"rl-environments"}],"trust":{"provenance":{"is_fork":false,"github_id":1278930806,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-07-28T12:00:49.145Z","maintenance":{"label":"Active","score":82,"methodology":"github_public_v1","releases_90d":0,"days_since_push":26,"last_release_at":null},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-11T11:59:47.797Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-07-28T12:00:49.785Z"},"license_spdx":{"value":"Other","source":"github.license","observed_at":"2026-07-28T12:00:49.785Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":null,"constraints":null,"when_to_use":["Need diverse resources encompassing papers, blogs, talks, tools, and benchmarks specifically curated for AI agent evaluation","Looking to understand how AI agents are evaluated in both LLMs and RL environments"],"when_not_to_use":["Require real-time interactive support or direct tool integrations not covered by a static resource list","Seeking proprietary tools from specific vendors rather than open resources and community content"],"source":"enrich:decision_facts","observed_at":"2026-07-17T05:08:17.351Z"},"constraint_facets":null,"decision_summary":[{"label":"Adopt for","value":"Curated resources for AI agent evaluation with BenchFlow backing its maintenance"}]}}