{"data":{"node":{"slug":"evalplus-evalplus","name":"evalplus","tagline":"Rigorous evaluation of LLM-synthesized code","github_url":"https://github.com/evalplus/evalplus","owner":"evalplus","repo":"evalplus","owner_avatar_url":"https://avatars.githubusercontent.com/u/132106461?v=4","primary_language":"Python","stars":1794,"forks":205,"topics":["benchmark","chatgpt","efficiency","gpt-4","large-language-models","program-synthesis","testing"],"archived":false,"github_pushed_at":"2025-10-02T22:56:38+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/evalplus-evalplus","markdown_url":"https://www.graphcanon.com/tools/evalplus-evalplus.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evalplus-evalplus","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evalplus-evalplus"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"benchmark","name":"benchmark"},{"slug":"chatgpt","name":"chatgpt"},{"slug":"efficiency","name":"efficiency"},{"slug":"program-synthesis","name":"program-synthesis"},{"slug":"testing","name":"testing"}],"edges":[],"neighbours":[{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":17226,"forks":1736,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-07-27T11:33:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"zai-org-codegeex","name":"CodeGeeX","tagline":"CodeGeeX is an open multilingual code generation model implemented in Mindspore and available via PyTorch.","github_url":"https://github.com/zai-org/CodeGeeX","owner":"zai-org","repo":"CodeGeeX","owner_avatar_url":"https://avatars.githubusercontent.com/u/223098841?v=4","primary_language":"Python","stars":8809,"forks":688,"topics":["code-generation","pretrained-models","tools"],"archived":false,"github_pushed_at":"2024-08-13T05:59:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/zai-org-codegeex","markdown_url":"https://www.graphcanon.com/tools/zai-org-codegeex.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zai-org-codegeex","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zai-org-codegeex","shared_categories":[]},{"slug":"openai-human-eval","name":"human-eval","tagline":"Evaluating Large Language Models Trained on Code","github_url":"https://github.com/openai/human-eval","owner":"openai","repo":"human-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/14957082?v=4","primary_language":"Python","stars":3331,"forks":452,"topics":[],"archived":false,"github_pushed_at":"2025-01-17T18:22:17+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/openai-human-eval","markdown_url":"https://www.graphcanon.com/tools/openai-human-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openai-human-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openai-human-eval","shared_categories":["evaluation-observability"]},{"slug":"code-yeongyu-lazycodex","name":"lazycodex","tagline":"Agent harness for complex codebases with project memory and execution planning","github_url":"https://github.com/code-yeongyu/lazycodex","owner":"code-yeongyu","repo":"lazycodex","owner_avatar_url":"https://avatars.githubusercontent.com/u/11153873?v=4","primary_language":"TypeScript","stars":3189,"forks":198,"topics":["ai","ai-agents","claude","claude-code","cli","codex","developer-tools","lazy","lazycodex","oh-my-openagent","omo","openai","orchestration","typescript"],"archived":false,"github_pushed_at":"2026-08-09T08:51:31+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/code-yeongyu-lazycodex","markdown_url":"https://www.graphcanon.com/tools/code-yeongyu-lazycodex.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/code-yeongyu-lazycodex","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=code-yeongyu-lazycodex","shared_categories":[]},{"slug":"salesforce-codet5","name":"CodeT5","tagline":"Home of CodeT5: Open Code LLMs for Code Understanding and Generation","github_url":"https://github.com/salesforce/CodeT5","owner":"salesforce","repo":"CodeT5","owner_avatar_url":"https://avatars.githubusercontent.com/u/453694?v=4","primary_language":"Python","stars":3098,"forks":487,"topics":["code-generation","code-intelligence","code-understanding","language-model","large-language-models"],"archived":true,"github_pushed_at":"2026-06-25T16:27:18+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/salesforce-codet5","markdown_url":"https://www.graphcanon.com/tools/salesforce-codet5.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/salesforce-codet5","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=salesforce-codet5","shared_categories":[]},{"slug":"rlancemartin-auto-evaluator","name":"auto-evaluator","tagline":"A lightweight evaluation tool for question-answering using Langchain","github_url":"https://github.com/rlancemartin/auto-evaluator","owner":"rlancemartin","repo":"auto-evaluator","owner_avatar_url":"https://avatars.githubusercontent.com/u/122662504?v=4","primary_language":"Python","stars":1105,"forks":92,"topics":[],"archived":false,"github_pushed_at":"2023-05-10T02:00:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator","markdown_url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rlancemartin-auto-evaluator","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rlancemartin-auto-evaluator","shared_categories":["evaluation-observability"]},{"slug":"bigcode-project-bigcode-evaluation-harness","name":"bigcode-evaluation-harness","tagline":"A framework for evaluating autoregressive code generation language models.","github_url":"https://github.com/bigcode-project/bigcode-evaluation-harness","owner":"bigcode-project","repo":"bigcode-evaluation-harness","owner_avatar_url":"https://avatars.githubusercontent.com/u/110470554?v=4","primary_language":"Python","stars":1055,"forks":261,"topics":[],"archived":false,"github_pushed_at":"2025-07-22T13:18:09+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/bigcode-project-bigcode-evaluation-harness","markdown_url":"https://www.graphcanon.com/tools/bigcode-project-bigcode-evaluation-harness.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bigcode-project-bigcode-evaluation-harness","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bigcode-project-bigcode-evaluation-harness","shared_categories":["evaluation-observability"]},{"slug":"livecodebench-livecodebench","name":"LiveCodeBench","tagline":"Holistic and contamination-free evaluation of large language models for code","github_url":"https://github.com/LiveCodeBench/LiveCodeBench","owner":"LiveCodeBench","repo":"LiveCodeBench","owner_avatar_url":"https://avatars.githubusercontent.com/u/161278213?v=4","primary_language":"Python","stars":925,"forks":195,"topics":["code-execution","code-generation","code-llms","code-repair","gpt-4","test-generation"],"archived":false,"github_pushed_at":"2025-07-16T00:58:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/livecodebench-livecodebench","markdown_url":"https://www.graphcanon.com/tools/livecodebench-livecodebench.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/livecodebench-livecodebench","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=livecodebench-livecodebench","shared_categories":["evaluation-observability"]},{"slug":"floridsleeves-llmdebugger","name":"LLMDebugger","tagline":"A Large Language Model Debugger verifying runtime execution step by step","github_url":"https://github.com/FloridSleeves/LLMDebugger","owner":"FloridSleeves","repo":"LLMDebugger","owner_avatar_url":"https://avatars.githubusercontent.com/u/23695653?v=4","primary_language":"Python","stars":587,"forks":56,"topics":[],"archived":false,"github_pushed_at":"2024-09-10T23:32:12+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/floridsleeves-llmdebugger","markdown_url":"https://www.graphcanon.com/tools/floridsleeves-llmdebugger.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/floridsleeves-llmdebugger","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=floridsleeves-llmdebugger","shared_categories":["evaluation-observability"]},{"slug":"declare-lab-instruct-eval","name":"instruct-eval","tagline":"Quantitative evaluation for instruction-tuned language models","github_url":"https://github.com/declare-lab/instruct-eval","owner":"declare-lab","repo":"instruct-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/59164695?v=4","primary_language":"Python","stars":552,"forks":45,"topics":["instruct-tuning","llm"],"archived":false,"github_pushed_at":"2024-03-10T05:00:00+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval","markdown_url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/declare-lab-instruct-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=declare-lab-instruct-eval","shared_categories":["evaluation-observability"]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]},{"slug":"alopatenko-llmevaluation","name":"LLMEvaluation","tagline":"A comprehensive guide to LLM evaluation methods","github_url":"https://github.com/alopatenko/LLMEvaluation","owner":"alopatenko","repo":"LLMEvaluation","owner_avatar_url":"https://avatars.githubusercontent.com/u/7122933?v=4","primary_language":"HTML","stars":196,"forks":22,"topics":["evaluation","generative-ai-benchmarking","llm","llm-benchmarking","llm-evaluation"],"archived":false,"github_pushed_at":"2026-07-06T01:17:36+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation","markdown_url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alopatenko-llmevaluation","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alopatenko-llmevaluation","shared_categories":["evaluation-observability"]}]}}