{"data":{"node":{"slug":"kolenaio-autoarena","name":"autoarena","tagline":"Automated evaluation of LLMs and RAG systems","github_url":"https://github.com/kolenaIO/autoarena","owner":"kolenaIO","repo":"autoarena","owner_avatar_url":"https://avatars.githubusercontent.com/u/77010818?v=4","primary_language":"TypeScript","stars":108,"forks":9,"topics":["ai","evaluation","hacktoberfest","llm","llm-evaluation","rag","testing"],"archived":false,"github_pushed_at":"2024-12-16T12:25:44+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/kolenaio-autoarena","markdown_url":"https://www.graphcanon.com/tools/kolenaio-autoarena.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kolenaio-autoarena","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kolenaio-autoarena"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"ai","name":"ai"},{"slug":"evaluation","name":"evaluation"},{"slug":"llm-evaluation","name":"llm-evaluation"},{"slug":"rag","name":"rag"},{"slug":"testing","name":"testing"}],"edges":[],"neighbours":[{"slug":"patchy631-ai-engineering-hub","name":"ai-engineering-hub","tagline":"Tutorials on LLMs, RAGs, and real-world AI agent applications","github_url":"https://github.com/patchy631/ai-engineering-hub","owner":"patchy631","repo":"ai-engineering-hub","owner_avatar_url":"https://avatars.githubusercontent.com/u/38653995?v=4","primary_language":"Jupyter Notebook","stars":37020,"forks":6107,"topics":["agents","ai","llms","machine-learning","mcp","rag"],"archived":false,"github_pushed_at":"2026-07-27T18:43:06+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/patchy631-ai-engineering-hub","markdown_url":"https://www.graphcanon.com/tools/patchy631-ai-engineering-hub.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/patchy631-ai-engineering-hub","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=patchy631-ai-engineering-hub","shared_categories":[]},{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":17226,"forks":1736,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-07-27T11:33:31+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"eleutherai-lm-evaluation-harness","name":"lm-evaluation-harness","tagline":"A framework for few-shot evaluation of language models.","github_url":"https://github.com/EleutherAI/lm-evaluation-harness","owner":"EleutherAI","repo":"lm-evaluation-harness","owner_avatar_url":"https://avatars.githubusercontent.com/u/68924597?v=4","primary_language":"Python","stars":13560,"forks":3467,"topics":["evaluation-framework","language-model","transformer"],"archived":false,"github_pushed_at":"2026-07-13T20:18:15+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness","markdown_url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eleutherai-lm-evaluation-harness","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eleutherai-lm-evaluation-harness","shared_categories":["evaluation-observability"]},{"slug":"arindam200-awesome-ai-apps","name":"awesome-ai-apps","tagline":"A curated list of AI applications showcasing RAG, agents, and workflows.","github_url":"https://github.com/Arindam200/awesome-ai-apps","owner":"Arindam200","repo":"awesome-ai-apps","owner_avatar_url":"https://avatars.githubusercontent.com/u/109217591?v=4","primary_language":"Python","stars":13494,"forks":1760,"topics":["agents","ai","hacktoberfest","llm","mcp"],"archived":false,"github_pushed_at":"2026-08-19T05:04:11+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/arindam200-awesome-ai-apps","markdown_url":"https://www.graphcanon.com/tools/arindam200-awesome-ai-apps.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/arindam200-awesome-ai-apps","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=arindam200-awesome-ai-apps","shared_categories":[]},{"slug":"giskard-ai-giskard-oss","name":"giskard-oss","tagline":"Open-Source Evaluation & Testing library for LLM Agents","github_url":"https://github.com/Giskard-AI/giskard-oss","owner":"Giskard-AI","repo":"giskard-oss","owner_avatar_url":"https://avatars.githubusercontent.com/u/71782571?v=4","primary_language":"Python","stars":5727,"forks":511,"topics":["agent-evaluation","ai-red-team","ai-security","ai-testing","fairness-ai","llm","llm-eval","llm-evaluation","llm-security","llmops","ml-testing","ml-validation","mlops","rag-evaluation","red-team-tools","responsible-ai","trustworthy-ai"],"archived":false,"github_pushed_at":"2026-08-01T23:22:37+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/giskard-ai-giskard-oss","markdown_url":"https://www.graphcanon.com/tools/giskard-ai-giskard-oss.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/giskard-ai-giskard-oss","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=giskard-ai-giskard-oss","shared_categories":["evaluation-observability"]},{"slug":"hegelai-prompttools","name":"prompttools","tagline":"Open-source tools for prompt testing and experimentation","github_url":"https://github.com/hegelai/prompttools","owner":"hegelai","repo":"prompttools","owner_avatar_url":"https://avatars.githubusercontent.com/u/136523567?v=4","primary_language":"Python","stars":3046,"forks":255,"topics":["deep-learning","developer-tools","embeddings","large-language-models","llms","machine-learning","prompt-engineering","python","vector-search"],"archived":false,"github_pushed_at":"2026-02-11T03:24:04+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/hegelai-prompttools","markdown_url":"https://www.graphcanon.com/tools/hegelai-prompttools.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/hegelai-prompttools","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=hegelai-prompttools","shared_categories":[]},{"slug":"benchflow-ai-awesome-evals","name":"awesome-evals","tagline":"A curated library of resources for building and evaluating AI agents","github_url":"https://github.com/benchflow-ai/awesome-evals","owner":"benchflow-ai","repo":"awesome-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/190338344?v=4","primary_language":null,"stars":761,"forks":71,"topics":["agent-evaluation","ai-agents","awesome","awesome-list","benchmarks","evals","llm","llm-evaluation","rl-environments"],"archived":false,"github_pushed_at":"2026-07-01T22:53:19+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals","markdown_url":"https://www.graphcanon.com/tools/benchflow-ai-awesome-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/benchflow-ai-awesome-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=benchflow-ai-awesome-evals","shared_categories":["evaluation-observability"]},{"slug":"declare-lab-instruct-eval","name":"instruct-eval","tagline":"Quantitative evaluation for instruction-tuned language models","github_url":"https://github.com/declare-lab/instruct-eval","owner":"declare-lab","repo":"instruct-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/59164695?v=4","primary_language":"Python","stars":552,"forks":45,"topics":["instruct-tuning","llm"],"archived":false,"github_pushed_at":"2024-03-10T05:00:00+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval","markdown_url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/declare-lab-instruct-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=declare-lab-instruct-eval","shared_categories":["evaluation-observability"]},{"slug":"rhesis-ai-rhesis","name":"rhesis","tagline":"Testing platform for AI teams to generate tests and evaluate system performance","github_url":"https://github.com/rhesis-ai/rhesis","owner":"rhesis-ai","repo":"rhesis","owner_avatar_url":"https://avatars.githubusercontent.com/u/168341335?v=4","primary_language":"Python","stars":381,"forks":31,"topics":["annotations","feedback-loop","hypothesis-testing","llmops","regression-testing","systematic-evaluation"],"archived":false,"github_pushed_at":"2026-07-28T15:13:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/rhesis-ai-rhesis","markdown_url":"https://www.graphcanon.com/tools/rhesis-ai-rhesis.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rhesis-ai-rhesis","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rhesis-ai-rhesis","shared_categories":["evaluation-observability"]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]},{"slug":"emirsahin1-llm-axe","name":"llm-axe","tagline":"Toolkit for quick implementation of LLM powered applications","github_url":"https://github.com/emirsahin1/llm-axe","owner":"emirsahin1","repo":"llm-axe","owner_avatar_url":"https://avatars.githubusercontent.com/u/50391065?v=4","primary_language":"Python","stars":275,"forks":38,"topics":["function-calling","llama3","llm","local-llm","ollama","pdf-llm"],"archived":false,"github_pushed_at":"2025-01-05T19:47:01+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/emirsahin1-llm-axe","markdown_url":"https://www.graphcanon.com/tools/emirsahin1-llm-axe.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/emirsahin1-llm-axe","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=emirsahin1-llm-axe","shared_categories":[]},{"slug":"alopatenko-llmevaluation","name":"LLMEvaluation","tagline":"A comprehensive guide to LLM evaluation methods","github_url":"https://github.com/alopatenko/LLMEvaluation","owner":"alopatenko","repo":"LLMEvaluation","owner_avatar_url":"https://avatars.githubusercontent.com/u/7122933?v=4","primary_language":"HTML","stars":196,"forks":22,"topics":["evaluation","generative-ai-benchmarking","llm","llm-benchmarking","llm-evaluation"],"archived":false,"github_pushed_at":"2026-07-06T01:17:36+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation","markdown_url":"https://www.graphcanon.com/tools/alopatenko-llmevaluation.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alopatenko-llmevaluation","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alopatenko-llmevaluation","shared_categories":["evaluation-observability"]}]}}