{"data":{"node":{"slug":"zli12321-qa-metrics","name":"qa_metrics","tagline":"A Python package for basic QA evaluations of large language models.","github_url":"https://github.com/zli12321/qa_metrics","owner":"zli12321","repo":"qa_metrics","owner_avatar_url":"https://avatars.githubusercontent.com/u/60415163?v=4","primary_language":"Python","stars":64,"forks":6,"topics":["exact-matching","llm","llm-evaluation","llm-evaluation-framework","llm-evaluation-toolkit","qa-automation-test","reward-modeling","rl-training"],"archived":false,"github_pushed_at":"2025-07-18T22:42:40+00:00","maintenance_label":"Dormant","stars_delta_30d":2,"url":"https://www.graphcanon.com/tools/zli12321-qa-metrics","markdown_url":"https://www.graphcanon.com/tools/zli12321-qa-metrics.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zli12321-qa-metrics","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zli12321-qa-metrics"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"exact-matching","name":"exact-matching"},{"slug":"llm-evaluation","name":"llm-evaluation"},{"slug":"qa-automation-test","name":"qa-automation-test"}],"edges":[],"neighbours":[{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":18342,"forks":1953,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-09-18T17:06:58+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"eleutherai-lm-evaluation-harness","name":"lm-evaluation-harness","tagline":"A framework for few-shot evaluation of language models.","github_url":"https://github.com/EleutherAI/lm-evaluation-harness","owner":"EleutherAI","repo":"lm-evaluation-harness","owner_avatar_url":"https://avatars.githubusercontent.com/u/68924597?v=4","primary_language":"Python","stars":13906,"forks":3547,"topics":["evaluation-framework","language-model","transformer"],"archived":false,"github_pushed_at":"2026-09-01T13:51:29+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness","markdown_url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eleutherai-lm-evaluation-harness","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eleutherai-lm-evaluation-harness","shared_categories":["evaluation-observability"]},{"slug":"evolvinglmms-lab-lmms-eval","name":"lmms-eval","tagline":"One-for-All Multimodal Evaluation Toolkit Across Text, Image, Video, and Audio Tasks","github_url":"https://github.com/EvolvingLMMs-Lab/lmms-eval","owner":"EvolvingLMMs-Lab","repo":"lmms-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/154951679?v=4","primary_language":"Python","stars":4368,"forks":639,"topics":["agi","audio-evaluation","benchmark","evaluation","large-language-models","llm-evaluation","multimodal","multimodal-evaluation","video-understanding","vision-language-model","vlm"],"archived":false,"github_pushed_at":"2026-08-06T02:22:23+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/evolvinglmms-lab-lmms-eval","markdown_url":"https://www.graphcanon.com/tools/evolvinglmms-lab-lmms-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evolvinglmms-lab-lmms-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evolvinglmms-lab-lmms-eval","shared_categories":["evaluation-observability"]},{"slug":"hegelai-prompttools","name":"prompttools","tagline":"Open-source tools for prompt testing and experimentation","github_url":"https://github.com/hegelai/prompttools","owner":"hegelai","repo":"prompttools","owner_avatar_url":"https://avatars.githubusercontent.com/u/136523567?v=4","primary_language":"Python","stars":3055,"forks":256,"topics":["deep-learning","developer-tools","embeddings","large-language-models","llms","machine-learning","prompt-engineering","python","vector-search"],"archived":false,"github_pushed_at":"2026-02-11T03:24:04+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/hegelai-prompttools","markdown_url":"https://www.graphcanon.com/tools/hegelai-prompttools.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/hegelai-prompttools","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=hegelai-prompttools","shared_categories":[]},{"slug":"rlancemartin-auto-evaluator","name":"auto-evaluator","tagline":"A lightweight evaluation tool for question-answering using Langchain","github_url":"https://github.com/rlancemartin/auto-evaluator","owner":"rlancemartin","repo":"auto-evaluator","owner_avatar_url":"https://avatars.githubusercontent.com/u/122662504?v=4","primary_language":"Python","stars":1102,"forks":92,"topics":[],"archived":false,"github_pushed_at":"2023-05-10T02:00:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator","markdown_url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rlancemartin-auto-evaluator","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rlancemartin-auto-evaluator","shared_categories":["evaluation-observability"]},{"slug":"georgian-io-llm-finetuning-toolkit","name":"LLM-Finetuning-Toolkit","tagline":"Toolkit for fine-tuning and testing open-source large language models","github_url":"https://github.com/georgian-io/LLM-Finetuning-Toolkit","owner":"georgian-io","repo":"LLM-Finetuning-Toolkit","owner_avatar_url":"https://avatars.githubusercontent.com/u/10764713?v=4","primary_language":"Python","stars":870,"forks":107,"topics":["ablation-study","classification","falcon","fine-tuning","finetuning","flan-t5","large-language-models","llama2","llm-test","lora","mistral-7b","nlp","nlp-machine-learning","qlora","redpajama","summarization","unit-testing","zephyr"],"archived":false,"github_pushed_at":"2026-05-04T16:33:40+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/georgian-io-llm-finetuning-toolkit","markdown_url":"https://www.graphcanon.com/tools/georgian-io-llm-finetuning-toolkit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/georgian-io-llm-finetuning-toolkit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=georgian-io-llm-finetuning-toolkit","shared_categories":[]},{"slug":"declare-lab-instruct-eval","name":"instruct-eval","tagline":"Quantitative evaluation for instruction-tuned language models","github_url":"https://github.com/declare-lab/instruct-eval","owner":"declare-lab","repo":"instruct-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/59164695?v=4","primary_language":"Python","stars":553,"forks":45,"topics":["instruct-tuning","llm"],"archived":false,"github_pushed_at":"2024-03-10T05:00:00+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval","markdown_url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/declare-lab-instruct-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=declare-lab-instruct-eval","shared_categories":["evaluation-observability"]},{"slug":"curated-awesome-lists-awesome-llms-fine-tuning","name":"awesome-llms-fine-tuning","tagline":"A comprehensive collection of resources for fine-tuning Large Language Models.","github_url":"https://github.com/Curated-Awesome-Lists/awesome-llms-fine-tuning","owner":"Curated-Awesome-Lists","repo":"awesome-llms-fine-tuning","owner_avatar_url":"https://avatars.githubusercontent.com/u/142611331?v=4","primary_language":null,"stars":527,"forks":80,"topics":["ai","awesome-list","deep-learning","fine-tuning","gpt","large-language-models","llms","machine-learning","nlp","transformers"],"archived":false,"github_pushed_at":"2026-09-04T19:22:27+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/curated-awesome-lists-awesome-llms-fine-tuning","markdown_url":"https://www.graphcanon.com/tools/curated-awesome-lists-awesome-llms-fine-tuning.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/curated-awesome-lists-awesome-llms-fine-tuning","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=curated-awesome-lists-awesome-llms-fine-tuning","shared_categories":[]},{"slug":"jonathanchaveztamales-llm-leaderboard","name":"llm-leaderboard","tagline":"Comprehensive LLM benchmark scores and provider prices","github_url":"https://github.com/JonathanChavezTamales/llm-leaderboard","owner":"JonathanChavezTamales","repo":"llm-leaderboard","owner_avatar_url":"https://avatars.githubusercontent.com/u/22694942?v=4","primary_language":"JavaScript","stars":356,"forks":40,"topics":["llm","llm-agents","llm-evaluation","llmops","llms-benchmarking"],"archived":false,"github_pushed_at":"2025-10-24T17:47:59+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/jonathanchaveztamales-llm-leaderboard","markdown_url":"https://www.graphcanon.com/tools/jonathanchaveztamales-llm-leaderboard.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jonathanchaveztamales-llm-leaderboard","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jonathanchaveztamales-llm-leaderboard","shared_categories":["evaluation-observability"]},{"slug":"verifywise-ai-verifywise","name":"verifywise","tagline":"Complete AI governance and LLM Evals platform","github_url":"https://github.com/verifywise-ai/verifywise","owner":"verifywise-ai","repo":"verifywise","owner_avatar_url":"https://avatars.githubusercontent.com/u/262239808?v=4","primary_language":"TypeScript","stars":354,"forks":117,"topics":["ai","ai-auditing","ai-compliance","ai-governance","ai-governance-model","ai-risk","audit","auditing","compliance","eu-ai-act","governance","grc","iso27001","iso42001","llm-eval","llm-evaluation","nist-ai-rmf","risk-management"],"archived":false,"github_pushed_at":"2026-09-19T19:03:23+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/verifywise-ai-verifywise","markdown_url":"https://www.graphcanon.com/tools/verifywise-ai-verifywise.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/verifywise-ai-verifywise","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=verifywise-ai-verifywise","shared_categories":["evaluation-observability"]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]},{"slug":"alopatenko-llmevaluation","name":"LLMEvaluation","tagline":"A comprehensive guide to LLM evaluation methods","github_url":"https://github.com/alopatenko/LLMEvaluation","owner":"alopatenko","repo":"LLMEvaluation","owner_avatar_url":"https://avatars.githubusercontent.com/u/7122933?v=4","primary_language":"HTML","stars":201,"forks":24,"topics":["evaluation","generative-ai-benchmarking","llm","llm-benchmarking","llm-evaluation"],"archived":false,"github_pushed_at":"2026-09-15T00:35:53+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation","markdown_url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alopatenko-llmevaluation","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alopatenko-llmevaluation","shared_categories":["evaluation-observability"]}]}}