{"data":{"node":{"slug":"isen-zhang-aclue","name":"ACLUE","tagline":"Evaluation Benchmark for Ancient Chinese Language Comprehension","github_url":"https://github.com/isen-zhang/ACLUE","owner":"isen-zhang","repo":"ACLUE","owner_avatar_url":"https://avatars.githubusercontent.com/u/86350285?v=4","primary_language":"Python","stars":34,"forks":0,"topics":[],"archived":false,"github_pushed_at":"2024-03-20T18:24:17+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/isen-zhang-aclue","markdown_url":"https://www.graphcanon.com/tools/isen-zhang-aclue.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/isen-zhang-aclue","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=isen-zhang-aclue"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"ancient-texts","name":"ancient texts"},{"slug":"chinese-language","name":"chinese language"},{"slug":"language-models-evaluation","name":"language models evaluation"},{"slug":"nlp-benchmarks","name":"nlp benchmarks"}],"edges":[],"neighbours":[{"slug":"nirdiamant-rag-techniques","name":"RAG_Techniques","tagline":"Showcases advanced techniques for Retrieval-Augmented Generation (RAG) systems with detailed notebook tutorials.","github_url":"https://github.com/NirDiamant/RAG_Techniques","owner":"NirDiamant","repo":"RAG_Techniques","owner_avatar_url":"https://avatars.githubusercontent.com/u/28316913?v=4","primary_language":"Jupyter Notebook","stars":29076,"forks":3540,"topics":["agentic-rag","ai","embeddings","generative-ai","gpt","langchain","llama-index","llm","llms","machine-learning","nlp","openai","python","rag","retrieval-augmented-generation","semantic-search","tutorials","vector-database"],"archived":false,"github_pushed_at":"2026-08-15T00:52:05+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/nirdiamant-rag-techniques","markdown_url":"https://www.graphcanon.com/tools/nirdiamant-rag-techniques.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nirdiamant-rag-techniques","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nirdiamant-rag-techniques","shared_categories":[]},{"slug":"aihubcn-awesome-chinese-llm","name":"Awesome-Chinese-LLM","tagline":"整理开源的中文大语言模型","github_url":"https://github.com/AiHubCN/Awesome-Chinese-LLM","owner":"AiHubCN","repo":"Awesome-Chinese-LLM","owner_avatar_url":"https://avatars.githubusercontent.com/u/29895268?v=4","primary_language":null,"stars":22738,"forks":2134,"topics":["awesome-lists","chatglm","chinese","llama","llm","nlp"],"archived":false,"github_pushed_at":"2026-05-10T05:03:06+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/aihubcn-awesome-chinese-llm","markdown_url":"https://www.graphcanon.com/tools/aihubcn-awesome-chinese-llm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/aihubcn-awesome-chinese-llm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=aihubcn-awesome-chinese-llm","shared_categories":[]},{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":17226,"forks":1736,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-07-27T11:33:31+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"eleutherai-lm-evaluation-harness","name":"lm-evaluation-harness","tagline":"A framework for few-shot evaluation of language models.","github_url":"https://github.com/EleutherAI/lm-evaluation-harness","owner":"EleutherAI","repo":"lm-evaluation-harness","owner_avatar_url":"https://avatars.githubusercontent.com/u/68924597?v=4","primary_language":"Python","stars":13560,"forks":3467,"topics":["evaluation-framework","language-model","transformer"],"archived":false,"github_pushed_at":"2026-07-13T20:18:15+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness","markdown_url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eleutherai-lm-evaluation-harness","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eleutherai-lm-evaluation-harness","shared_categories":["evaluation-observability"]},{"slug":"embeddings-benchmark-mteb","name":"mteb","tagline":"State-of-the-art evaluation of embeddings across languages and modalities","github_url":"https://github.com/embeddings-benchmark/mteb","owner":"embeddings-benchmark","repo":"mteb","owner_avatar_url":"https://avatars.githubusercontent.com/u/103029531?v=4","primary_language":"Python","stars":3400,"forks":670,"topics":["benchmark","bitext-mining","clustering","embeddings","evaluation","information-retrieval","low-resource-nlp","mteb","multilingual-nlp","multimodal","neural-search","reranking","retrieval","sbert","semantic-search","sentence-transformers","sts","text-classification","text-embedding"],"archived":false,"github_pushed_at":"2026-08-21T21:26:00+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/embeddings-benchmark-mteb","markdown_url":"https://www.graphcanon.com/tools/embeddings-benchmark-mteb.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/embeddings-benchmark-mteb","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=embeddings-benchmark-mteb","shared_categories":["evaluation-observability"]},{"slug":"franxyao-chain-of-thought-hub","name":"chain-of-thought-hub","tagline":"Benchmarking large language models' complex reasoning ability with chain-of-thought prompting","github_url":"https://github.com/FranxYao/chain-of-thought-hub","owner":"FranxYao","repo":"chain-of-thought-hub","owner_avatar_url":"https://avatars.githubusercontent.com/u/17723677?v=4","primary_language":"Jupyter Notebook","stars":2774,"forks":144,"topics":[],"archived":false,"github_pushed_at":"2024-08-04T09:40:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/franxyao-chain-of-thought-hub","markdown_url":"https://www.graphcanon.com/tools/franxyao-chain-of-thought-hub.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/franxyao-chain-of-thought-hub","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=franxyao-chain-of-thought-hub","shared_categories":["evaluation-observability"]},{"slug":"huangowen-awesome-llm-compression","name":"Awesome-LLM-Compression","tagline":"Awesome LLM compression research papers and tools to accelerate LLM training and inference.","github_url":"https://github.com/HuangOwen/Awesome-LLM-Compression","owner":"HuangOwen","repo":"Awesome-LLM-Compression","owner_avatar_url":"https://avatars.githubusercontent.com/u/24937399?v=4","primary_language":null,"stars":1859,"forks":129,"topics":[],"archived":false,"github_pushed_at":"2026-06-30T15:26:46+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression","markdown_url":"https://www.graphcanon.com/tools/huangowen-awesome-llm-compression.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huangowen-awesome-llm-compression","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huangowen-awesome-llm-compression","shared_categories":[]},{"slug":"livecodebench-livecodebench","name":"LiveCodeBench","tagline":"Holistic and contamination-free evaluation of large language models for code","github_url":"https://github.com/LiveCodeBench/LiveCodeBench","owner":"LiveCodeBench","repo":"LiveCodeBench","owner_avatar_url":"https://avatars.githubusercontent.com/u/161278213?v=4","primary_language":"Python","stars":925,"forks":195,"topics":["code-execution","code-generation","code-llms","code-repair","gpt-4","test-generation"],"archived":false,"github_pushed_at":"2025-07-16T00:58:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/livecodebench-livecodebench","markdown_url":"https://www.graphcanon.com/tools/livecodebench-livecodebench.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/livecodebench-livecodebench","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=livecodebench-livecodebench","shared_categories":["evaluation-observability"]},{"slug":"benchflow-ai-awesome-evals","name":"awesome-evals","tagline":"A curated library of resources for building and evaluating AI agents","github_url":"https://github.com/benchflow-ai/awesome-evals","owner":"benchflow-ai","repo":"awesome-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/190338344?v=4","primary_language":null,"stars":761,"forks":71,"topics":["agent-evaluation","ai-agents","awesome","awesome-list","benchmarks","evals","llm","llm-evaluation","rl-environments"],"archived":false,"github_pushed_at":"2026-07-01T22:53:19+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals","markdown_url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/benchflow-ai-awesome-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=benchflow-ai-awesome-evals","shared_categories":["evaluation-observability"]},{"slug":"declare-lab-instruct-eval","name":"instruct-eval","tagline":"Quantitative evaluation for instruction-tuned language models","github_url":"https://github.com/declare-lab/instruct-eval","owner":"declare-lab","repo":"instruct-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/59164695?v=4","primary_language":"Python","stars":552,"forks":45,"topics":["instruct-tuning","llm"],"archived":false,"github_pushed_at":"2024-03-10T05:00:00+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval","markdown_url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/declare-lab-instruct-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=declare-lab-instruct-eval","shared_categories":["evaluation-observability"]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]},{"slug":"alopatenko-llmevaluation","name":"LLMEvaluation","tagline":"A comprehensive guide to LLM evaluation methods","github_url":"https://github.com/alopatenko/LLMEvaluation","owner":"alopatenko","repo":"LLMEvaluation","owner_avatar_url":"https://avatars.githubusercontent.com/u/7122933?v=4","primary_language":"HTML","stars":196,"forks":22,"topics":["evaluation","generative-ai-benchmarking","llm","llm-benchmarking","llm-evaluation"],"archived":false,"github_pushed_at":"2026-07-06T01:17:36+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation","markdown_url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alopatenko-llmevaluation","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alopatenko-llmevaluation","shared_categories":["evaluation-observability"]}]}}