{"data":{"slug":"zli12321-qa-metrics","name":"qa_metrics","tagline":"A Python package for basic QA evaluations of large language models.","github_url":"https://github.com/zli12321/qa_metrics","owner":"zli12321","repo":"qa_metrics","owner_avatar_url":"https://avatars.githubusercontent.com/u/60415163?v=4","primary_language":"Python","stars":64,"forks":6,"topics":["exact-matching","llm","llm-evaluation","llm-evaluation-framework","llm-evaluation-toolkit","qa-automation-test","reward-modeling","rl-training"],"archived":false,"github_pushed_at":"2025-07-18T22:42:40+00:00","maintenance_label":"Dormant","stars_delta_30d":2,"url":"https://www.graphcanon.com/tools/zli12321-qa-metrics","markdown_url":"https://www.graphcanon.com/tools/zli12321-qa-metrics.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zli12321-qa-metrics","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zli12321-qa-metrics","description":"An easy python package to run quick basic QA evaluations. This package includes standardized QA evaluation metrics and semantic evaluation metrics: Black-box and Open-Source large language model prompting and evaluation, exact match, F1 Score, PEDANT semantic match, transformer match. Our package also supports prompting OPENAI and Anthropic API.","homepage_url":null,"license":"MIT","open_issues":0,"watchers":1,"ai_summary":"Provides standardized metrics and tools for evaluating LLMs","readme_excerpt":"### Installation\n```bash\npip install qa-metrics\n```\n\n---\n\n## 📝 License\n\nThis project is licensed under the [MIT License](LICENSE.md).","github_created_at":"2024-01-21T15:56:42+00:00","created_at":"2026-07-15T10:40:22.537524+00:00","updated_at":"2026-09-20T04:25:10.484787+00:00","categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"exact-matching","name":"exact-matching"},{"slug":"llm-evaluation","name":"llm-evaluation"},{"slug":"qa-automation-test","name":"qa-automation-test"}],"trust":{"provenance":{"is_fork":false,"github_id":746280574,"owner_type":"User","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-09-09T06:00:54.956Z","maintenance":{"label":"Dormant","score":18,"methodology":"github_public_v1","releases_90d":0,"days_since_push":417,"last_release_at":null,"stars_delta_30d":2,"open_issues_delta_30d":0},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-15T10:40:23.951Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-09-09T06:00:55.448Z"},"languages":{"value":["python"],"source":"github.language","observed_at":"2026-09-09T06:00:55.448Z"},"license_spdx":{"value":"MIT","source":"github.license","observed_at":"2026-09-09T06:00:55.448Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":null,"constraints":null,"when_to_use":["When you need to evaluate the performance of large language models with built-in standardized metrics like exact match and F1 Score.","If your project requires both black-box evaluation capabilities and access to APIs from major providers for model assessment."],"when_not_to_use":["Avoid if you seek advanced customization or fine-tuning options not present in qa_metrics for metric calculation methods beyond its provided set.","Not ideal when needing specific evaluation tools that are not Black-box or open-source models, as the package focuses on these types of evaluations primarily."],"source":"enrich:decision_facts","observed_at":"2026-07-16T20:34:58.991Z"},"constraint_facets":null,"decision_summary":[{"label":"Adopt for","value":"qa_metrics is a Python library for evaluating LLMs using standardized QA and semantic metrics, including support for Black-box and open-source models along with APIs from OpenAI and Anthropic."},{"label":"License detail","value":"MIT License allows for free use and distribution with attribution required by retaining the copyright notice and license text in any redistribution."}]}}