{"data":{"node":{"slug":"poloclub-llm-self-defense","name":"llm-self-defense","tagline":"LLM Self Defense: By Self Examination, LLMs know they are being tricked","github_url":"https://github.com/poloclub/llm-self-defense","owner":"poloclub","repo":"llm-self-defense","owner_avatar_url":"https://avatars.githubusercontent.com/u/19315506?v=4","primary_language":"Python","stars":52,"forks":7,"topics":[],"archived":false,"github_pushed_at":"2024-05-21T07:51:46+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/poloclub-llm-self-defense","markdown_url":"https://www.graphcanon.com/tools/poloclub-llm-self-defense.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/poloclub-llm-self-defense","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=poloclub-llm-self-defense"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"adversarial-prompts","name":"adversarial prompts"},{"slug":"gpt-3-5","name":"gpt 3.5"},{"slug":"harmful-content-reduction","name":"harmful content reduction"},{"slug":"llama-2","name":"llama-2"},{"slug":"llms","name":"llms"},{"slug":"safety-measures","name":"safety measures"},{"slug":"self-defense-mechanism","name":"self defense mechanism"}],"edges":[],"neighbours":[{"slug":"llm-attacks-llm-attacks","name":"llm-attacks","tagline":"Universal and Transferable Attacks on Aligned Language Models","github_url":"https://github.com/llm-attacks/llm-attacks","owner":"llm-attacks","repo":"llm-attacks","owner_avatar_url":"https://avatars.githubusercontent.com/u/140664770?v=4","primary_language":"Python","stars":4756,"forks":633,"topics":[],"archived":false,"github_pushed_at":"2024-08-02T06:02:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/llm-attacks-llm-attacks","markdown_url":"https://www.graphcanon.com/tools/llm-attacks-llm-attacks.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/llm-attacks-llm-attacks","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=llm-attacks-llm-attacks","shared_categories":["evaluation-observability"]},{"slug":"meta-llama-purplellama","name":"PurpleLlama","tagline":"Set of tools to assess and improve LLM security","github_url":"https://github.com/meta-llama/PurpleLlama","owner":"meta-llama","repo":"PurpleLlama","owner_avatar_url":"https://avatars.githubusercontent.com/u/153379578?v=4","primary_language":"Python","stars":4330,"forks":766,"topics":[],"archived":false,"github_pushed_at":"2026-07-27T21:53:59+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/meta-llama-purplellama","markdown_url":"https://www.graphcanon.com/tools/meta-llama-purplellama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/meta-llama-purplellama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=meta-llama-purplellama","shared_categories":["evaluation-observability"]},{"slug":"mirascope-mirascope","name":"mirascope","tagline":"The LLM Anti-Framework","github_url":"https://github.com/Mirascope/mirascope","owner":"Mirascope","repo":"mirascope","owner_avatar_url":"https://avatars.githubusercontent.com/u/153858533?v=4","primary_language":"Python","stars":1520,"forks":123,"topics":["artificial-intelligence","developer-tools","llm","llm-agent","llm-tools","python","typescript"],"archived":false,"github_pushed_at":"2026-07-29T12:45:01+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/mirascope-mirascope","markdown_url":"https://www.graphcanon.com/tools/mirascope-mirascope.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/mirascope-mirascope","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=mirascope-mirascope","shared_categories":[]},{"slug":"deadbits-vigil-llm","name":"vigil-llm","tagline":"Detect prompt injections and other risky inputs in LLMs","github_url":"https://github.com/deadbits/vigil-llm","owner":"deadbits","repo":"vigil-llm","owner_avatar_url":"https://avatars.githubusercontent.com/u/1332757?v=4","primary_language":"Python","stars":496,"forks":56,"topics":["adversarial-attacks","adversarial-machine-learning","large-language-models","llm-security","llmops","prompt-injection","security-tools","yara-scanner"],"archived":false,"github_pushed_at":"2024-01-31T18:43:41+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/deadbits-vigil-llm","markdown_url":"https://www.graphcanon.com/tools/deadbits-vigil-llm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/deadbits-vigil-llm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=deadbits-vigil-llm","shared_categories":["evaluation-observability"]},{"slug":"liu00222-open-prompt-injection","name":"Open-Prompt-Injection","tagline":"Benchmark and toolkit for prompt injection attacks and defenses in LLMs","github_url":"https://github.com/liu00222/Open-Prompt-Injection","owner":"liu00222","repo":"Open-Prompt-Injection","owner_avatar_url":"https://avatars.githubusercontent.com/u/42081599?v=4","primary_language":"Python","stars":470,"forks":74,"topics":["llm","llm-security","llms","prompt-injection","prompt-injection-tool","security-and-privacy"],"archived":false,"github_pushed_at":"2025-10-29T17:11:34+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/liu00222-open-prompt-injection","markdown_url":"https://www.graphcanon.com/tools/liu00222-open-prompt-injection.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/liu00222-open-prompt-injection","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=liu00222-open-prompt-injection","shared_categories":["evaluation-observability"]},{"slug":"llm-tuning-safety-llms-finetuning-safety","name":"LLMs-Finetuning-Safety","tagline":"Demonstrates safety risks in fine-tuning GPT-3.5 Turbo with adversarial examples","github_url":"https://github.com/LLM-Tuning-Safety/LLMs-Finetuning-Safety","owner":"LLM-Tuning-Safety","repo":"LLMs-Finetuning-Safety","owner_avatar_url":"https://avatars.githubusercontent.com/u/146881603?v=4","primary_language":"Python","stars":358,"forks":38,"topics":["alignment","llm","llm-finetuning"],"archived":false,"github_pushed_at":"2024-02-23T21:19:44+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/llm-tuning-safety-llms-finetuning-safety","markdown_url":"https://www.graphcanon.com/tools/llm-tuning-safety-llms-finetuning-safety.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/llm-tuning-safety-llms-finetuning-safety","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=llm-tuning-safety-llms-finetuning-safety","shared_categories":["evaluation-observability"]},{"slug":"libr-ai-do-not-answer","name":"do-not-answer","tagline":"A Dataset for Evaluating Safeguards in LLMs","github_url":"https://github.com/Libr-AI/do-not-answer","owner":"Libr-AI","repo":"do-not-answer","owner_avatar_url":"https://avatars.githubusercontent.com/u/133515165?v=4","primary_language":"Jupyter Notebook","stars":339,"forks":29,"topics":[],"archived":false,"github_pushed_at":"2024-06-07T14:55:51+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/libr-ai-do-not-answer","markdown_url":"https://www.graphcanon.com/tools/libr-ai-do-not-answer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/libr-ai-do-not-answer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=libr-ai-do-not-answer","shared_categories":["evaluation-observability"]},{"slug":"jagilley-fact-checker","name":"fact-checker","tagline":"Fact-checking LLM outputs with self-ask","github_url":"https://github.com/jagilley/fact-checker","owner":"jagilley","repo":"fact-checker","owner_avatar_url":"https://avatars.githubusercontent.com/u/37783831?v=4","primary_language":"Jupyter Notebook","stars":313,"forks":39,"topics":["llm","python"],"archived":false,"github_pushed_at":"2023-10-23T21:08:13+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/jagilley-fact-checker","markdown_url":"https://www.graphcanon.com/tools/jagilley-fact-checker.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jagilley-fact-checker","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jagilley-fact-checker","shared_categories":["evaluation-observability"]},{"slug":"alopatenko-llmevaluation","name":"LLMEvaluation","tagline":"A comprehensive guide to LLM evaluation methods","github_url":"https://github.com/alopatenko/LLMEvaluation","owner":"alopatenko","repo":"LLMEvaluation","owner_avatar_url":"https://avatars.githubusercontent.com/u/7122933?v=4","primary_language":"HTML","stars":196,"forks":22,"topics":["evaluation","generative-ai-benchmarking","llm","llm-benchmarking","llm-evaluation"],"archived":false,"github_pushed_at":"2026-07-06T01:17:36+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation","markdown_url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alopatenko-llmevaluation","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alopatenko-llmevaluation","shared_categories":["evaluation-observability"]},{"slug":"zenguard-ai-fast-llm-security-guardrails","name":"fast-llm-security-guardrails","tagline":"The fastest Trust Layer for AI Agents","github_url":"https://github.com/ZenGuard-AI/fast-llm-security-guardrails","owner":"ZenGuard-AI","repo":"fast-llm-security-guardrails","owner_avatar_url":"https://avatars.githubusercontent.com/u/163059052?v=4","primary_language":"Python","stars":154,"forks":21,"topics":["agentic-ai","ai-agent","ai-agents","ai-runtime","cx-agent","llm-guard","llm-guardrails","llm-privacy","llm-security","prompt-security","security"],"archived":false,"github_pushed_at":"2026-02-03T18:31:21+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/zenguard-ai-fast-llm-security-guardrails","markdown_url":"https://www.graphcanon.com/tools/zenguard-ai-fast-llm-security-guardrails.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zenguard-ai-fast-llm-security-guardrails","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zenguard-ai-fast-llm-security-guardrails","shared_categories":["evaluation-observability"]},{"slug":"safellama-plexiglass","name":"plexiglass","tagline":"A toolkit for detecting and protecting against vulnerabilities in Large Language Models (LLMs).","github_url":"https://github.com/safellama/plexiglass","owner":"safellama","repo":"plexiglass","owner_avatar_url":"https://avatars.githubusercontent.com/u/141878442?v=4","primary_language":"Python","stars":153,"forks":18,"topics":["adversarial-attacks","adversarial-machine-learning","cybersecurity","deep-learning","deep-neural-networks","machine-learning","security"],"archived":false,"github_pushed_at":"2026-02-04T22:18:58+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/safellama-plexiglass","markdown_url":"https://www.graphcanon.com/tools/safellama-plexiglass.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/safellama-plexiglass","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=safellama-plexiglass","shared_categories":["evaluation-observability"]},{"slug":"microsoft-bipia","name":"BIPIA","tagline":"Benchmark for evaluating LLM robustness to indirect prompt injection attacks.","github_url":"https://github.com/microsoft/BIPIA","owner":"microsoft","repo":"BIPIA","owner_avatar_url":"https://avatars.githubusercontent.com/u/6154722?v=4","primary_language":"Python","stars":149,"forks":19,"topics":["llm-security"],"archived":false,"github_pushed_at":"2024-04-15T02:08:17+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/microsoft-bipia","markdown_url":"https://www.graphcanon.com/tools/microsoft-bipia.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/microsoft-bipia","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=microsoft-bipia","shared_categories":["evaluation-observability"]}]}}