{"data":{"node":{"slug":"open-compass-deveval","name":"DevEval","tagline":"A Comprehensive Benchmark for Software Development","github_url":"https://github.com/open-compass/DevEval","owner":"open-compass","repo":"DevEval","owner_avatar_url":"https://avatars.githubusercontent.com/u/143521324?v=4","primary_language":"Python","stars":138,"forks":13,"topics":[],"archived":false,"github_pushed_at":"2024-05-30T13:10:52+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/open-compass-deveval","markdown_url":"https://www.graphcanon.com/tools/open-compass-deveval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/open-compass-deveval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=open-compass-deveval"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"benchmark","name":"benchmark"},{"slug":"docker-supported","name":"docker-supported"},{"slug":"python","name":"python"},{"slug":"software-development","name":"software-development"}],"edges":[],"neighbours":[{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":17226,"forks":1736,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-07-27T11:33:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"swe-bench-swe-bench","name":"SWE-bench","tagline":"Benchmark for assessing language models' capability to resolve real-world Github issues","github_url":"https://github.com/SWE-bench/SWE-bench","owner":"SWE-bench","repo":"SWE-bench","owner_avatar_url":"https://avatars.githubusercontent.com/u/139597579?v=4","primary_language":"Python","stars":5576,"forks":930,"topics":["benchmark","language-model","software-engineering"],"archived":false,"github_pushed_at":"2026-07-27T05:34:27+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/swe-bench-swe-bench","markdown_url":"https://www.graphcanon.com/tools/swe-bench-swe-bench.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/swe-bench-swe-bench","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=swe-bench-swe-bench","shared_categories":["evaluation-observability"]},{"slug":"stanford-crfm-helm","name":"helm","tagline":"Holistic, reproducible and transparent evaluation of foundation models","github_url":"https://github.com/stanford-crfm/helm","owner":"stanford-crfm","repo":"helm","owner_avatar_url":"https://avatars.githubusercontent.com/u/75054807?v=4","primary_language":"Python","stars":2873,"forks":406,"topics":[],"archived":false,"github_pushed_at":"2026-08-01T01:23:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/stanford-crfm-helm","markdown_url":"https://www.graphcanon.com/tools/stanford-crfm-helm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/stanford-crfm-helm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=stanford-crfm-helm","shared_categories":["evaluation-observability"]},{"slug":"evalplus-evalplus","name":"evalplus","tagline":"Rigorous evaluation of LLM-synthesized code","github_url":"https://github.com/evalplus/evalplus","owner":"evalplus","repo":"evalplus","owner_avatar_url":"https://avatars.githubusercontent.com/u/132106461?v=4","primary_language":"Python","stars":1794,"forks":205,"topics":["benchmark","chatgpt","efficiency","gpt-4","large-language-models","program-synthesis","testing"],"archived":false,"github_pushed_at":"2025-10-02T22:56:38+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/evalplus-evalplus","markdown_url":"https://www.graphcanon.com/tools/evalplus-evalplus.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evalplus-evalplus","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evalplus-evalplus","shared_categories":["evaluation-observability"]},{"slug":"bigcode-project-bigcode-evaluation-harness","name":"bigcode-evaluation-harness","tagline":"A framework for evaluating autoregressive code generation language models.","github_url":"https://github.com/bigcode-project/bigcode-evaluation-harness","owner":"bigcode-project","repo":"bigcode-evaluation-harness","owner_avatar_url":"https://avatars.githubusercontent.com/u/110470554?v=4","primary_language":"Python","stars":1055,"forks":261,"topics":[],"archived":false,"github_pushed_at":"2025-07-22T13:18:09+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/bigcode-project-bigcode-evaluation-harness","markdown_url":"https://www.graphcanon.com/tools/bigcode-project-bigcode-evaluation-harness.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bigcode-project-bigcode-evaluation-harness","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bigcode-project-bigcode-evaluation-harness","shared_categories":["evaluation-observability"]},{"slug":"livecodebench-livecodebench","name":"LiveCodeBench","tagline":"Holistic and contamination-free evaluation of large language models for code","github_url":"https://github.com/LiveCodeBench/LiveCodeBench","owner":"LiveCodeBench","repo":"LiveCodeBench","owner_avatar_url":"https://avatars.githubusercontent.com/u/161278213?v=4","primary_language":"Python","stars":925,"forks":195,"topics":["code-execution","code-generation","code-llms","code-repair","gpt-4","test-generation"],"archived":false,"github_pushed_at":"2025-07-16T00:58:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/livecodebench-livecodebench","markdown_url":"https://www.graphcanon.com/tools/livecodebench-livecodebench.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/livecodebench-livecodebench","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=livecodebench-livecodebench","shared_categories":["evaluation-observability"]},{"slug":"nuprl-multipl-e","name":"MultiPL-E","tagline":"A multi-programming language benchmark for LLMs","github_url":"https://github.com/nuprl/MultiPL-E","owner":"nuprl","repo":"MultiPL-E","owner_avatar_url":"https://avatars.githubusercontent.com/u/4161665?v=4","primary_language":"Python","stars":313,"forks":57,"topics":[],"archived":false,"github_pushed_at":"2026-04-12T16:59:02+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/nuprl-multipl-e","markdown_url":"https://www.graphcanon.com/tools/nuprl-multipl-e.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nuprl-multipl-e","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nuprl-multipl-e","shared_categories":["evaluation-observability"]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]},{"slug":"xlang-ai-ds-1000","name":"DS-1000","tagline":"Benchmark and code for evaluating large language models on data science tasks","github_url":"https://github.com/xlang-ai/DS-1000","owner":"xlang-ai","repo":"DS-1000","owner_avatar_url":"https://avatars.githubusercontent.com/u/128829376?v=4","primary_language":"Python","stars":276,"forks":31,"topics":["benchmark","code-generation","data-science","large-language-models","semantic-parsing"],"archived":false,"github_pushed_at":"2024-10-30T17:43:46+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/xlang-ai-ds-1000","markdown_url":"https://www.graphcanon.com/tools/xlang-ai-ds-1000.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/xlang-ai-ds-1000","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=xlang-ai-ds-1000","shared_categories":[]},{"slug":"leolty-repobench","name":"repobench","tagline":"Benchmarking Repository-Level Code Auto-Completion Systems","github_url":"https://github.com/Leolty/repobench","owner":"Leolty","repo":"repobench","owner_avatar_url":"https://avatars.githubusercontent.com/u/58247268?v=4","primary_language":"Python","stars":214,"forks":13,"topics":[],"archived":false,"github_pushed_at":"2024-08-16T07:08:32+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/leolty-repobench","markdown_url":"https://www.graphcanon.com/tools/leolty-repobench.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/leolty-repobench","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=leolty-repobench","shared_categories":[]},{"slug":"zeno-ml-zeno","name":"zeno","tagline":"AI Data Management & Evaluation Platform","github_url":"https://github.com/zeno-ml/zeno","owner":"zeno-ml","repo":"zeno","owner_avatar_url":"https://avatars.githubusercontent.com/u/109821189?v=4","primary_language":"Svelte","stars":214,"forks":11,"topics":["ai","data-science","evaluation","evaluation-framework","machine-learning","python"],"archived":true,"github_pushed_at":"2023-10-05T19:02:16+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/zeno-ml-zeno","markdown_url":"https://www.graphcanon.com/tools/zeno-ml-zeno.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zeno-ml-zeno","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zeno-ml-zeno","shared_categories":["evaluation-observability"]},{"slug":"alopatenko-llmevaluation","name":"LLMEvaluation","tagline":"A comprehensive guide to LLM evaluation methods","github_url":"https://github.com/alopatenko/LLMEvaluation","owner":"alopatenko","repo":"LLMEvaluation","owner_avatar_url":"https://avatars.githubusercontent.com/u/7122933?v=4","primary_language":"HTML","stars":196,"forks":22,"topics":["evaluation","generative-ai-benchmarking","llm","llm-benchmarking","llm-evaluation"],"archived":false,"github_pushed_at":"2026-07-06T01:17:36+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation","markdown_url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alopatenko-llmevaluation","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alopatenko-llmevaluation","shared_categories":["evaluation-observability"]}]}}