{"data":{"slug":"allenai-commongen-eval","name":"CommonGen-Eval","tagline":"Evaluating LLMs with CommonGen-Lite","github_url":"https://github.com/allenai/CommonGen-Eval","owner":"allenai","repo":"CommonGen-Eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/5667695?v=4","primary_language":"Python","stars":95,"forks":3,"topics":["chatgpt","evaluation","gpt-evaluation","llama2","llm","llm-evaluation","text-generation"],"archived":false,"github_pushed_at":"2024-03-21T00:28:45+00:00","maintenance_label":"Dormant","stars_delta_30d":0,"url":"https://www.graphcanon.com/tools/allenai-commongen-eval","markdown_url":"https://www.graphcanon.com/tools/allenai-commongen-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/allenai-commongen-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=allenai-commongen-eval","description":"Evaluating LLMs with CommonGen-Lite","homepage_url":"https://inklab.usc.edu/CommonGen/","license":"Apache-2.0","open_issues":1,"watchers":3,"ai_summary":"This tool evaluates large language models using the CommonGen-Lite dataset.","readme_excerpt":"## Installation \n\n```bash \npip install -r requirements.txt\npython -m spacy download en_core_web_lg\n```","github_created_at":"2024-01-04T03:32:09+00:00","created_at":"2026-07-15T10:39:05.468435+00:00","updated_at":"2026-09-20T04:24:46.290085+00:00","categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"evaluation","name":"evaluation"},{"slug":"llm-evaluation","name":"llm-evaluation"}],"trust":{"provenance":{"is_fork":false,"github_id":738787172,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-09-08T06:01:32.004Z","maintenance":{"label":"Dormant","score":18,"methodology":"github_public_v1","releases_90d":0,"days_since_push":901,"last_release_at":null,"stars_delta_30d":0,"open_issues_delta_30d":0},"security_summary":{"status":"findings","scanner":"osv@v1","low_count":94,"high_count":0,"last_scan_at":"2026-07-15T10:39:06.935Z","medium_count":0,"scan_profile":"deps","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-09-08T06:01:32.437Z"},"languages":{"value":["python"],"source":"github.language","observed_at":"2026-09-08T06:01:32.437Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-09-08T06:01:32.437Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":{"notes":["Install Python dependencies using `pip install -r requirements.txt`","Download necessary Spacy models with `python -m spacy download en_core_web_lg`"],"min_ram_gb":null,"requires_docker":false},"constraints":{"min_ram_gb":null,"requires_docker":false},"when_to_use":["Use CommonGen-Eval when you need to assess how well an LLM can generate a diverse set of common-sense facts or statements based on given concepts.","Select it specifically for testing text-generation capabilities aligned with everyday language understanding and generation."],"when_not_to_use":["Avoid using CommonGen-Eval if your evaluation priorities align more closely with task-specific benchmarks outside of general-language diversification.","Do not use this tool if your project requires an evaluation framework that focuses heavily on the ability to answer specific factual questions or handle domain-specific language."],"source":"enrich:decision_facts","observed_at":"2026-07-16T20:02:07.538Z"},"constraint_facets":{"min_ram_gb":null,"requires_docker":false},"decision_summary":[{"label":"Requirements","value":"Install Python dependencies using `pip install -r requirements.txt`; Download necessary Spacy models with `python -m spacy download en_core_web_lg`"},{"label":"Adopt for","value":"CommonGen-Eval is designed to evaluate large language models using the CommonGen-Lite dataset, focusing on generating diverse phrases and sentences."}]}}