{"data":{"node":{"slug":"jeinlee1991-chinese-llm-benchmark","name":"chinese-llm-benchmark","tagline":"ReLE评测：中文AI大模型能力评测","github_url":"https://github.com/jeinlee1991/chinese-llm-benchmark","owner":"jeinlee1991","repo":"chinese-llm-benchmark","owner_avatar_url":"https://avatars.githubusercontent.com/u/46815718?v=4","primary_language":null,"stars":6353,"forks":261,"topics":["agentic-ai","artificial-intelligence","llm-agent","llm-evaluation"],"archived":false,"github_pushed_at":"2026-08-04T04:30:01+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jeinlee1991-chinese-llm-benchmark","markdown_url":"https://www.graphcanon.com/tools/jeinlee1991-chinese-llm-benchmark.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jeinlee1991-chinese-llm-benchmark","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jeinlee1991-chinese-llm-benchmark"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"agentic-ai","name":"agentic-ai"},{"slug":"artificial-intelligence","name":"artificial-intelligence"},{"slug":"llm-agent","name":"llm-agent"},{"slug":"llm-evaluation","name":"llm-evaluation"}],"edges":[],"neighbours":[{"slug":"aihubcn-awesome-chinese-llm","name":"Awesome-Chinese-LLM","tagline":"整理开源的中文大语言模型","github_url":"https://github.com/AiHubCN/Awesome-Chinese-LLM","owner":"AiHubCN","repo":"Awesome-Chinese-LLM","owner_avatar_url":"https://avatars.githubusercontent.com/u/29895268?v=4","primary_language":null,"stars":22738,"forks":2134,"topics":["awesome-lists","chatglm","chinese","llama","llm","nlp"],"archived":false,"github_pushed_at":"2026-05-10T05:03:06+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/aihubcn-awesome-chinese-llm","markdown_url":"https://www.graphcanon.com/tools/aihubcn-awesome-chinese-llm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/aihubcn-awesome-chinese-llm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=aihubcn-awesome-chinese-llm","shared_categories":[]},{"slug":"openai-evals","name":"evals","tagline":"Framework for evaluating LLMs and LLM systems with an open-source registry of benchmarks.","github_url":"https://github.com/openai/evals","owner":"openai","repo":"evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/14957082?v=4","primary_language":"Python","stars":19127,"forks":3050,"topics":[],"archived":false,"github_pushed_at":"2026-04-14T15:29:57+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/openai-evals","markdown_url":"https://www.graphcanon.com/tools/openai-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openai-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openai-evals","shared_categories":["evaluation-observability"]},{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":17226,"forks":1736,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-07-27T11:33:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"stanford-crfm-helm","name":"helm","tagline":"Holistic, reproducible and transparent evaluation of foundation models","github_url":"https://github.com/stanford-crfm/helm","owner":"stanford-crfm","repo":"helm","owner_avatar_url":"https://avatars.githubusercontent.com/u/75054807?v=4","primary_language":"Python","stars":2873,"forks":406,"topics":[],"archived":false,"github_pushed_at":"2026-08-01T01:23:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/stanford-crfm-helm","markdown_url":"https://www.graphcanon.com/tools/stanford-crfm-helm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/stanford-crfm-helm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=stanford-crfm-helm","shared_categories":["evaluation-observability"]},{"slug":"huggingface-lighteval","name":"lighteval","tagline":"All-in-one toolkit for evaluating LLMs across multiple backends","github_url":"https://github.com/huggingface/lighteval","owner":"huggingface","repo":"lighteval","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":2508,"forks":523,"topics":["evaluation","evaluation-framework","evaluation-metrics","huggingface"],"archived":false,"github_pushed_at":"2026-06-29T13:03:33+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huggingface-lighteval","markdown_url":"https://www.graphcanon.com/tools/huggingface-lighteval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-lighteval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-lighteval","shared_categories":["evaluation-observability"]},{"slug":"onejune2018-awesome-llm-eval","name":"Awesome-LLM-Eval","tagline":"Curated list for evaluation of large language models","github_url":"https://github.com/onejune2018/Awesome-LLM-Eval","owner":"onejune2018","repo":"Awesome-LLM-Eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/35061855?v=4","primary_language":null,"stars":654,"forks":82,"topics":["awsome-list","awsome-lists","benchmark","bert","chatglm","chatgpt","dataset","evaluation","gpt3","large-language-model","leaderboard","llama","llm","llm-evaluation","machine-learning","nlp","openai","qwen","rag"],"archived":false,"github_pushed_at":"2025-11-24T01:59:12+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/onejune2018-awesome-llm-eval","markdown_url":"https://www.graphcanon.com/tools/onejune2018-awesome-llm-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/onejune2018-awesome-llm-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=onejune2018-awesome-llm-eval","shared_categories":["evaluation-observability"]},{"slug":"alopatenko-llmevaluation","name":"LLMEvaluation","tagline":"A comprehensive guide to LLM evaluation methods","github_url":"https://github.com/alopatenko/LLMEvaluation","owner":"alopatenko","repo":"LLMEvaluation","owner_avatar_url":"https://avatars.githubusercontent.com/u/7122933?v=4","primary_language":"HTML","stars":196,"forks":22,"topics":["evaluation","generative-ai-benchmarking","llm","llm-benchmarking","llm-evaluation"],"archived":false,"github_pushed_at":"2026-07-06T01:17:36+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation","markdown_url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alopatenko-llmevaluation","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alopatenko-llmevaluation","shared_categories":["evaluation-observability"]},{"slug":"langwatch-langevals","name":"langevals","tagline":"Provides a platform for evaluating and benchmarking LLM models using various evaluators","github_url":"https://github.com/langwatch/langevals","owner":"langwatch","repo":"langevals","owner_avatar_url":"https://avatars.githubusercontent.com/u/146763322?v=4","primary_language":null,"stars":72,"forks":9,"topics":["evaluation","guardrails","llm","openai"],"archived":false,"github_pushed_at":"2026-02-15T18:31:16+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/langwatch-langevals","markdown_url":"https://www.graphcanon.com/tools/langwatch-langevals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langwatch-langevals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langwatch-langevals","shared_categories":["evaluation-observability"]}]}}