{"data":{"node":{"slug":"google-big-bench","name":"BIG-bench","tagline":"Collaborative benchmark for language model capabilities","github_url":"https://github.com/google/BIG-bench","owner":"google","repo":"BIG-bench","owner_avatar_url":"https://avatars.githubusercontent.com/u/1342004?v=4","primary_language":"Python","stars":3249,"forks":617,"topics":[],"archived":true,"github_pushed_at":"2024-07-19T11:57:37+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/google-big-bench","markdown_url":"https://www.graphcanon.com/tools/google-big-bench.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/google-big-bench","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=google-big-bench"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"benchmarking","name":"benchmarking"},{"slug":"evaluation","name":"evaluation"},{"slug":"language-models","name":"language-models"},{"slug":"seqio","name":"seqio"},{"slug":"t5x","name":"t5x"},{"slug":"tasks-creation","name":"tasks creation"}],"edges":[],"neighbours":[{"slug":"bradyfu-awesome-multimodal-large-language-models","name":"Awesome-Multimodal-Large-Language-Models","tagline":"Latest Advances on Multimodal Large Language Models","github_url":"https://github.com/BradyFU/Awesome-Multimodal-Large-Language-Models","owner":"BradyFU","repo":"Awesome-Multimodal-Large-Language-Models","owner_avatar_url":"https://avatars.githubusercontent.com/u/54254631?v=4","primary_language":null,"stars":17978,"forks":1133,"topics":["chain-of-thought","in-context-learning","instruction-following","instruction-tuning","large-language-models","large-vision-language-model","large-vision-language-models","multi-modality","multimodal-chain-of-thought","multimodal-in-context-learning","multimodal-instruction-tuning","multimodal-large-language-models","visual-instruction-tuning"],"archived":false,"github_pushed_at":"2026-08-14T17:17:50+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/bradyfu-awesome-multimodal-large-language-models","markdown_url":"https://www.graphcanon.com/tools/bradyfu-awesome-multimodal-large-language-models.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/bradyfu-awesome-multimodal-large-language-models","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=bradyfu-awesome-multimodal-large-language-models","shared_categories":["evaluation-observability"]},{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":17226,"forks":1736,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-07-27T11:33:31+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"eleutherai-lm-evaluation-harness","name":"lm-evaluation-harness","tagline":"A framework for few-shot evaluation of language models.","github_url":"https://github.com/EleutherAI/lm-evaluation-harness","owner":"EleutherAI","repo":"lm-evaluation-harness","owner_avatar_url":"https://avatars.githubusercontent.com/u/68924597?v=4","primary_language":"Python","stars":13560,"forks":3467,"topics":["evaluation-framework","language-model","transformer"],"archived":false,"github_pushed_at":"2026-07-13T20:18:15+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness","markdown_url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eleutherai-lm-evaluation-harness","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eleutherai-lm-evaluation-harness","shared_categories":["evaluation-observability"]},{"slug":"shishirpatil-gorilla","name":"gorilla","tagline":"Training and Evaluating LLMs for Function Calls (Tool Calls)","github_url":"https://github.com/ShishirPatil/gorilla","owner":"ShishirPatil","repo":"gorilla","owner_avatar_url":"https://avatars.githubusercontent.com/u/30296397?v=4","primary_language":"Python","stars":12988,"forks":1397,"topics":["api","api-documentation","chatgpt","claude-api","gpt-4-api","llm","openai-api","openai-functions"],"archived":false,"github_pushed_at":"2026-04-13T03:19:45+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/shishirpatil-gorilla","markdown_url":"https://www.graphcanon.com/tools/shishirpatil-gorilla.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/shishirpatil-gorilla","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=shishirpatil-gorilla","shared_categories":["evaluation-observability"]},{"slug":"rucaibox-llmsurvey","name":"LLMSurvey","tagline":"A comprehensive collection of papers and resources related to Large Language Models.","github_url":"https://github.com/RUCAIBox/LLMSurvey","owner":"RUCAIBox","repo":"LLMSurvey","owner_avatar_url":"https://avatars.githubusercontent.com/u/54706620?v=4","primary_language":"Python","stars":12205,"forks":931,"topics":["chain-of-thought","chatgpt","in-context-learning","instruction-tuning","large-language-models","llm","llms","natural-language-processing","pre-trained-language-models","pre-training","rlhf"],"archived":false,"github_pushed_at":"2025-03-11T09:51:42+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/rucaibox-llmsurvey","markdown_url":"https://www.graphcanon.com/tools/rucaibox-llmsurvey.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rucaibox-llmsurvey","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rucaibox-llmsurvey","shared_categories":["evaluation-observability"]},{"slug":"evolvinglmms-lab-lmms-eval","name":"lmms-eval","tagline":"One-for-All Multimodal Evaluation Toolkit Across Text, Image, Video, and Audio Tasks","github_url":"https://github.com/EvolvingLMMs-Lab/lmms-eval","owner":"EvolvingLMMs-Lab","repo":"lmms-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/154951679?v=4","primary_language":"Python","stars":4368,"forks":639,"topics":["agi","audio-evaluation","benchmark","evaluation","large-language-models","llm-evaluation","multimodal","multimodal-evaluation","video-understanding","vision-language-model","vlm"],"archived":false,"github_pushed_at":"2026-08-06T02:22:23+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/evolvinglmms-lab-lmms-eval","markdown_url":"https://www.graphcanon.com/tools/evolvinglmms-lab-lmms-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evolvinglmms-lab-lmms-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evolvinglmms-lab-lmms-eval","shared_categories":["evaluation-observability"]},{"slug":"embeddings-benchmark-mteb","name":"mteb","tagline":"State-of-the-art evaluation of embeddings across languages and modalities","github_url":"https://github.com/embeddings-benchmark/mteb","owner":"embeddings-benchmark","repo":"mteb","owner_avatar_url":"https://avatars.githubusercontent.com/u/103029531?v=4","primary_language":"Python","stars":3400,"forks":670,"topics":["benchmark","bitext-mining","clustering","embeddings","evaluation","information-retrieval","low-resource-nlp","mteb","multilingual-nlp","multimodal","neural-search","reranking","retrieval","sbert","semantic-search","sentence-transformers","sts","text-classification","text-embedding"],"archived":false,"github_pushed_at":"2026-08-21T21:26:00+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/embeddings-benchmark-mteb","markdown_url":"https://www.graphcanon.com/tools/embeddings-benchmark-mteb.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/embeddings-benchmark-mteb","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=embeddings-benchmark-mteb","shared_categories":["evaluation-observability"]},{"slug":"stanford-crfm-helm","name":"helm","tagline":"Holistic, reproducible and transparent evaluation of foundation models","github_url":"https://github.com/stanford-crfm/helm","owner":"stanford-crfm","repo":"helm","owner_avatar_url":"https://avatars.githubusercontent.com/u/75054807?v=4","primary_language":"Python","stars":2873,"forks":406,"topics":[],"archived":false,"github_pushed_at":"2026-08-01T01:23:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/stanford-crfm-helm","markdown_url":"https://www.graphcanon.com/tools/stanford-crfm-helm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/stanford-crfm-helm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=stanford-crfm-helm","shared_categories":["evaluation-observability"]},{"slug":"franxyao-chain-of-thought-hub","name":"chain-of-thought-hub","tagline":"Benchmarking large language models' complex reasoning ability with chain-of-thought prompting","github_url":"https://github.com/FranxYao/chain-of-thought-hub","owner":"FranxYao","repo":"chain-of-thought-hub","owner_avatar_url":"https://avatars.githubusercontent.com/u/17723677?v=4","primary_language":"Jupyter Notebook","stars":2774,"forks":144,"topics":[],"archived":false,"github_pushed_at":"2024-08-04T09:40:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/franxyao-chain-of-thought-hub","markdown_url":"https://www.graphcanon.com/tools/franxyao-chain-of-thought-hub.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/franxyao-chain-of-thought-hub","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=franxyao-chain-of-thought-hub","shared_categories":["evaluation-observability"]},{"slug":"livecodebench-livecodebench","name":"LiveCodeBench","tagline":"Holistic and contamination-free evaluation of large language models for code","github_url":"https://github.com/LiveCodeBench/LiveCodeBench","owner":"LiveCodeBench","repo":"LiveCodeBench","owner_avatar_url":"https://avatars.githubusercontent.com/u/161278213?v=4","primary_language":"Python","stars":925,"forks":195,"topics":["code-execution","code-generation","code-llms","code-repair","gpt-4","test-generation"],"archived":false,"github_pushed_at":"2025-07-16T00:58:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/livecodebench-livecodebench","markdown_url":"https://www.graphcanon.com/tools/livecodebench-livecodebench.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/livecodebench-livecodebench","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=livecodebench-livecodebench","shared_categories":["evaluation-observability"]},{"slug":"jailbreakbench-jailbreakbench","name":"jailbreakbench","tagline":"An Open Robustness Benchmark for Jailbreaking Language Models","github_url":"https://github.com/JailbreakBench/jailbreakbench","owner":"JailbreakBench","repo":"jailbreakbench","owner_avatar_url":"https://avatars.githubusercontent.com/u/151564101?v=4","primary_language":"Python","stars":645,"forks":75,"topics":[],"archived":false,"github_pushed_at":"2025-04-04T11:30:46+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/jailbreakbench-jailbreakbench","markdown_url":"https://www.graphcanon.com/tools/jailbreakbench-jailbreakbench.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jailbreakbench-jailbreakbench","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jailbreakbench-jailbreakbench","shared_categories":["evaluation-observability"]},{"slug":"declare-lab-instruct-eval","name":"instruct-eval","tagline":"Quantitative evaluation for instruction-tuned language models","github_url":"https://github.com/declare-lab/instruct-eval","owner":"declare-lab","repo":"instruct-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/59164695?v=4","primary_language":"Python","stars":552,"forks":45,"topics":["instruct-tuning","llm"],"archived":false,"github_pushed_at":"2024-03-10T05:00:00+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval","markdown_url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/declare-lab-instruct-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=declare-lab-instruct-eval","shared_categories":["evaluation-observability"]}]}}