{"data":{"node":{"slug":"hithink-research-gage","name":"GAGE","tagline":"Unified Evaluation Engine for AI Models","github_url":"https://github.com/HiThink-Research/GAGE","owner":"HiThink-Research","repo":"GAGE","owner_avatar_url":"https://avatars.githubusercontent.com/u/186997494?v=4","primary_language":"Python","stars":52,"forks":8,"topics":["agent","game-arena","llm","llm-evaluation","mllm-evaluation","sandbox-environment"],"archived":false,"github_pushed_at":"2026-06-02T06:49:47+00:00","maintenance_label":"Slowing","stars_delta_30d":1,"url":"https://www.graphcanon.com/tools/hithink-research-gage","markdown_url":"https://www.graphcanon.com/tools/hithink-research-gage.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/hithink-research-gage","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=hithink-research-gage"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"agents","name":"agents"},{"slug":"audio-models","name":"audio_models"},{"slug":"diffusion-models","name":"diffusion-models"},{"slug":"evaluation","name":"evaluation"},{"slug":"game-environments","name":"game_environments"},{"slug":"large-language-models","name":"large-language-models"},{"slug":"multimodal-models","name":"multimodal_models"},{"slug":"unified-engine","name":"unified_engine"}],"edges":[],"neighbours":[{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":18342,"forks":1953,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-09-18T17:06:58+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"raga-ai-hub-ragaai-catalyst","name":"RagaAI-Catalyst","tagline":"Python SDK for AI agent observability and evaluation","github_url":"https://github.com/raga-ai-hub/RagaAI-Catalyst","owner":"raga-ai-hub","repo":"RagaAI-Catalyst","owner_avatar_url":"https://avatars.githubusercontent.com/u/161833182?v=4","primary_language":"Python","stars":16162,"forks":3567,"topics":["agentic-ai","agentic-ai-development","agentneo","agents","ai-agent-monitoring","ai-application-debugging","ai-evaluation-tools","ai-performance-optimization","ai-tool-interaction-monitoring","llm-testing","llm-tracing","llmops"],"archived":false,"github_pushed_at":"2026-02-11T14:43:33+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/raga-ai-hub-ragaai-catalyst","markdown_url":"https://www.graphcanon.com/tools/raga-ai-hub-ragaai-catalyst.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/raga-ai-hub-ragaai-catalyst","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=raga-ai-hub-ragaai-catalyst","shared_categories":["evaluation-observability"]},{"slug":"eleutherai-lm-evaluation-harness","name":"lm-evaluation-harness","tagline":"A framework for few-shot evaluation of language models.","github_url":"https://github.com/EleutherAI/lm-evaluation-harness","owner":"EleutherAI","repo":"lm-evaluation-harness","owner_avatar_url":"https://avatars.githubusercontent.com/u/68924597?v=4","primary_language":"Python","stars":13906,"forks":3547,"topics":["evaluation-framework","language-model","transformer"],"archived":false,"github_pushed_at":"2026-09-01T13:51:29+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness","markdown_url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eleutherai-lm-evaluation-harness","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eleutherai-lm-evaluation-harness","shared_categories":["evaluation-observability"]},{"slug":"shishirpatil-gorilla","name":"gorilla","tagline":"Training and Evaluating LLMs for Function Calls (Tool Calls)","github_url":"https://github.com/ShishirPatil/gorilla","owner":"ShishirPatil","repo":"gorilla","owner_avatar_url":"https://avatars.githubusercontent.com/u/30296397?v=4","primary_language":"Python","stars":13017,"forks":1406,"topics":["api","api-documentation","chatgpt","claude-api","gpt-4-api","llm","openai-api","openai-functions"],"archived":false,"github_pushed_at":"2026-04-13T03:19:45+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/shishirpatil-gorilla","markdown_url":"https://www.graphcanon.com/tools/shishirpatil-gorilla.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/shishirpatil-gorilla","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=shishirpatil-gorilla","shared_categories":["evaluation-observability"]},{"slug":"evolvinglmms-lab-lmms-eval","name":"lmms-eval","tagline":"One-for-All Multimodal Evaluation Toolkit Across Text, Image, Video, and Audio Tasks","github_url":"https://github.com/EvolvingLMMs-Lab/lmms-eval","owner":"EvolvingLMMs-Lab","repo":"lmms-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/154951679?v=4","primary_language":"Python","stars":4368,"forks":639,"topics":["agi","audio-evaluation","benchmark","evaluation","large-language-models","llm-evaluation","multimodal","multimodal-evaluation","video-understanding","vision-language-model","vlm"],"archived":false,"github_pushed_at":"2026-08-06T02:22:23+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/evolvinglmms-lab-lmms-eval","markdown_url":"https://www.graphcanon.com/tools/evolvinglmms-lab-lmms-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evolvinglmms-lab-lmms-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evolvinglmms-lab-lmms-eval","shared_categories":["evaluation-observability"]},{"slug":"stanford-crfm-helm","name":"helm","tagline":"Holistic, reproducible and transparent evaluation of foundation models","github_url":"https://github.com/stanford-crfm/helm","owner":"stanford-crfm","repo":"helm","owner_avatar_url":"https://avatars.githubusercontent.com/u/75054807?v=4","primary_language":"Python","stars":2902,"forks":411,"topics":[],"archived":false,"github_pushed_at":"2026-09-01T01:33:19+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/stanford-crfm-helm","markdown_url":"https://www.graphcanon.com/tools/stanford-crfm-helm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/stanford-crfm-helm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=stanford-crfm-helm","shared_categories":["evaluation-observability"]},{"slug":"huggingface-lighteval","name":"lighteval","tagline":"All-in-one toolkit for evaluating LLMs across multiple backends","github_url":"https://github.com/huggingface/lighteval","owner":"huggingface","repo":"lighteval","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":2535,"forks":552,"topics":["evaluation","evaluation-framework","evaluation-metrics","huggingface"],"archived":false,"github_pushed_at":"2026-08-11T13:10:37+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huggingface-lighteval","markdown_url":"https://www.graphcanon.com/tools/huggingface-lighteval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-lighteval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-lighteval","shared_categories":["evaluation-observability"]},{"slug":"future-agi-future-agi","name":"future-agi","tagline":"Open-source, end-to-end platform for evaluating, observing, and improving LLM and AI agent applications","github_url":"https://github.com/future-agi/future-agi","owner":"future-agi","repo":"future-agi","owner_avatar_url":"https://avatars.githubusercontent.com/u/147392366?v=4","primary_language":"Python","stars":2032,"forks":627,"topics":["ai-agents","ai-evals","ai-gateway","ai-optimization","ai-simulations","evaluation-framework","guardrails","hallucination-detection","llm","llm-evaluation","llm-observability","llmops","model-evaluation","observability","opentelemetry","rag","rag-evaluation","simulation","telemetry","tracing"],"archived":false,"github_pushed_at":"2026-09-18T08:15:25+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/future-agi-future-agi","markdown_url":"https://www.graphcanon.com/tools/future-agi-future-agi.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/future-agi-future-agi","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=future-agi-future-agi","shared_categories":["evaluation-observability"]},{"slug":"declare-lab-instruct-eval","name":"instruct-eval","tagline":"Quantitative evaluation for instruction-tuned language models","github_url":"https://github.com/declare-lab/instruct-eval","owner":"declare-lab","repo":"instruct-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/59164695?v=4","primary_language":"Python","stars":553,"forks":45,"topics":["instruct-tuning","llm"],"archived":false,"github_pushed_at":"2024-03-10T05:00:00+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval","markdown_url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/declare-lab-instruct-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=declare-lab-instruct-eval","shared_categories":["evaluation-observability"]},{"slug":"verifywise-ai-verifywise","name":"verifywise","tagline":"Complete AI governance and LLM Evals platform","github_url":"https://github.com/verifywise-ai/verifywise","owner":"verifywise-ai","repo":"verifywise","owner_avatar_url":"https://avatars.githubusercontent.com/u/262239808?v=4","primary_language":"TypeScript","stars":354,"forks":117,"topics":["ai","ai-auditing","ai-compliance","ai-governance","ai-governance-model","ai-risk","audit","auditing","compliance","eu-ai-act","governance","grc","iso27001","iso42001","llm-eval","llm-evaluation","nist-ai-rmf","risk-management"],"archived":false,"github_pushed_at":"2026-09-19T19:03:23+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/verifywise-ai-verifywise","markdown_url":"https://www.graphcanon.com/tools/verifywise-ai-verifywise.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/verifywise-ai-verifywise","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=verifywise-ai-verifywise","shared_categories":["evaluation-observability"]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]},{"slug":"alopatenko-llmevaluation","name":"LLMEvaluation","tagline":"A comprehensive guide to LLM evaluation methods","github_url":"https://github.com/alopatenko/LLMEvaluation","owner":"alopatenko","repo":"LLMEvaluation","owner_avatar_url":"https://avatars.githubusercontent.com/u/7122933?v=4","primary_language":"HTML","stars":201,"forks":24,"topics":["evaluation","generative-ai-benchmarking","llm","llm-benchmarking","llm-evaluation"],"archived":false,"github_pushed_at":"2026-09-15T00:35:53+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation","markdown_url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alopatenko-llmevaluation","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alopatenko-llmevaluation","shared_categories":["evaluation-observability"]}]}}