{"data":{"slug":"whitecircle-circle-guard-bench","name":"circle-guard-bench","tagline":"AI benchmark for evaluating LLM guard systems","github_url":"https://github.com/whitecircle/circle-guard-bench","owner":"whitecircle","repo":"circle-guard-bench","owner_avatar_url":"https://avatars.githubusercontent.com/u/197516060?v=4","primary_language":"Python","stars":75,"forks":5,"topics":["ai","benchmark","benchmarking","guardrail","guardrails","jailbreak","large-language-model","large-language-models","llm","llm-as-a-judge","llm-eval","llm-evaluation","llm-jailbreaks","llm-security","safeguard"],"archived":false,"github_pushed_at":"2026-03-07T12:39:03+00:00","maintenance_label":"Slowing","stars_delta_30d":3,"url":"https://www.graphcanon.com/tools/whitecircle-circle-guard-bench","markdown_url":"https://www.graphcanon.com/tools/whitecircle-circle-guard-bench.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/whitecircle-circle-guard-bench","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=whitecircle-circle-guard-bench","description":"First-of-its-kind AI benchmark for evaluating the protection capabilities of large language model (LLM) guard systems (guardrails and safeguards)","homepage_url":"https://whitecircle.ai","license":"Apache-2.0","open_issues":1,"watchers":1,"ai_summary":"whitecircle/circle-guard-bench is the first-of-its-kind AI benchmark focused on evaluating protection capabilities of large language model (LLM) guard systems including guardrails and safeguards, featuring extensive evaluation scenarios.","readme_excerpt":"# Installation using Poetry (basic installation)\npoetry install\n\n---\n\n# Installation with additional inference engines\npoetry install --extras \"vllm sglang transformers\"\n\n---\n\n# Installation using pip (or uv)\npip install -e .\n\n---\n\n# Installation with additional inference engines using pip\npip install -e \".[vllm,sglang,transformers]\"\n```\n\n---\n\n## License\n\nThis project is licensed under the [Apache 2.0](LICENSE).","github_created_at":"2025-04-24T11:03:59+00:00","created_at":"2026-07-15T10:40:05.23382+00:00","updated_at":"2026-09-20T04:25:04.777156+00:00","categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"ai","name":"ai"},{"slug":"benchmarking","name":"benchmarking"},{"slug":"guardrail","name":"guardrail"},{"slug":"large-language-models","name":"large-language-models"},{"slug":"llm-evaluation","name":"llm-evaluation"},{"slug":"llm-security","name":"llm-security"}],"trust":{"provenance":{"is_fork":false,"github_id":971982531,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-09-09T06:00:44.233Z","maintenance":{"label":"Slowing","score":36,"methodology":"github_public_v1","releases_90d":0,"days_since_push":185,"last_release_at":null,"stars_delta_30d":3,"open_issues_delta_30d":1},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-15T10:40:06.785Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-09-09T06:00:44.723Z"},"has_cli":{"value":true,"source":"pyproject.toml:[project.scripts]","observed_at":"2026-09-09T06:00:44.723Z"},"languages":{"value":["python"],"source":"github.language+pyproject.toml","observed_at":"2026-09-09T06:00:44.723Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-09-09T06:00:44.723Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":{"min_ram_gb":null,"requires_docker":false},"constraints":{"min_ram_gb":null,"requires_docker":false},"when_to_use":["Use circle-guard-bench when you need to evaluate the effectiveness of guardrails and safeguards in your LLM environment, as it offers an unparalleled set of scenarios specific to these protections.","Consider using this tool if you are looking for detailed insights into how different guard systems perform under various potential security breaches or misuse attempts."],"when_not_to_use":["Avoid circle-guard-bench if your primary focus is on benchmarking the performance aspects like speed and latency of LLMs, as it specializes in evaluating protections rather than performance.","Do not use this tool when you intend to conduct general purpose evaluations or comparisons between different LLM models that do not specifically involve security-related guard systems."],"source":"enrich:decision_facts","observed_at":"2026-07-16T19:16:07.055Z"},"constraint_facets":{"min_ram_gb":null,"requires_docker":false},"decision_summary":[{"label":"Adopt for","value":"circle-guard-bench is a Python-based AI benchmark tool for evaluating large language model guard systems under various protection scenarios."}]}}