{"data":{"slug":"babelscape-alert","name":"ALERT","tagline":"A Comprehensive Benchmark for Assessing Large Language Models' Safety Through Red Teaming","github_url":"https://github.com/Babelscape/ALERT","owner":"Babelscape","repo":"ALERT","owner_avatar_url":"https://avatars.githubusercontent.com/u/90899893?v=4","primary_language":"Python","stars":59,"forks":8,"topics":["ai","artificial-intelligence","benchmark","bias-detection","llm","llm-evaluation","llm-safety","llm-safety-benchmark","nlp","nlp-machine-learning","red-teaming","safety-monitoring","transformers-models"],"archived":false,"github_pushed_at":"2024-09-20T08:29:57+00:00","maintenance_label":"Dormant","stars_delta_30d":0,"url":"https://www.graphcanon.com/tools/babelscape-alert","markdown_url":"https://www.graphcanon.com/tools/babelscape-alert.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/babelscape-alert","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=babelscape-alert","description":"Official repository for the paper \"ALERT: A Comprehensive Benchmark for Assessing Large Language Models’ Safety through Red Teaming\"","homepage_url":"https://arxiv.org/abs/2404.08676","license":"Other","open_issues":0,"watchers":2,"ai_summary":"This repository hosts the ALERT benchmark designed to evaluate large language models on safety through various testing categories including bias detection and red-teaming.","readme_excerpt":"# License\nAs specified in the paper, most of the prompts available in the ALERT benchmark are derived from the [Anthropic HH-RLHF dataset](https://github.com/anthropics/hh-rlhf/tree/master?tab=readme-ov-file) that is licensed under the MIT license. A copy of the license can be found [here](https://github.com/Babelscape/ALERT/blob/master/MIT_LICENSE). \n\nStarting from these prompts, we then employ a combination of keyword-matching and zero-shot classification strategies to filter out prompts that do not target one of our safety risk categories as well as to classify remaining ones. Furthermore, we designed templates to create new, additional prompts and provide sufficient support for each safety risk category in our benchmark. Finally, we adopt adversarial data augmentation methods to create the ALERT<sub>Adv</sub> subset of our benchmark. The ALERT benchmark is licensed under the CC BY-NC-SA 4.0 license. The text of the license can be found [here](https://github.com/Babelscape/ALERT/blob/master/LICENSE).\n\n<br>","github_created_at":"2024-04-06T11:01:51+00:00","created_at":"2026-07-15T10:40:26.228803+00:00","updated_at":"2026-09-20T04:25:11.768555+00:00","categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"ai","name":"ai"},{"slug":"artificial-intelligence","name":"artificial-intelligence"},{"slug":"benchmark","name":"benchmark"},{"slug":"bias-detection","name":"bias-detection"},{"slug":"llm-evaluation","name":"llm-evaluation"},{"slug":"llm-safety","name":"llm-safety"},{"slug":"nlp","name":"nlp"},{"slug":"transformers-models","name":"transformers-models"}],"trust":{"provenance":{"is_fork":false,"github_id":782897691,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-09-10T06:00:05.945Z","maintenance":{"label":"Dormant","score":18,"methodology":"github_public_v1","releases_90d":0,"days_since_push":719,"last_release_at":null,"stars_delta_30d":0,"open_issues_delta_30d":0},"security_summary":{"status":"ok","scanner":"osv@v1","low_count":0,"high_count":0,"last_scan_at":"2026-07-15T10:40:27.496Z","medium_count":0,"scan_profile":"deps","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-09-10T06:00:06.817Z"},"languages":{"value":["python"],"source":"github.language","observed_at":"2026-09-10T06:00:06.817Z"},"license_spdx":{"value":"Other","source":"github.license","observed_at":"2026-09-10T06:00:06.817Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":null,"constraints":null,"when_to_use":["When evaluating safety metrics of large language models through red-teaming approaches","For generating additional safety-focused prompts via template designs and keyword matching"],"when_not_to_use":["If your evaluation does not require bias detection or safety assessment under adversarial conditions","In scenarios where a broader range of model aspects beyond safety is needed, as ALERT focuses primarily on safety benchmarks"],"source":"enrich:decision_facts","observed_at":"2026-07-16T19:15:04.226Z"},"constraint_facets":null,"decision_summary":[{"label":"Adopt for","value":"ALERT is designed specifically for red-teaming based safety evaluation on large language models, using MIT licensed prompts and adversarial augmentation."}]}}