{"data":{"node":{"slug":"parea-ai-parea-sdk-py","name":"parea-sdk-py","tagline":"Python SDK for experimenting, testing, evaluating and monitoring LLM-powered applications.","github_url":"https://github.com/parea-ai/parea-sdk-py","owner":"parea-ai","repo":"parea-sdk-py","owner_avatar_url":"https://avatars.githubusercontent.com/u/135163833?v=4","primary_language":"Python","stars":82,"forks":13,"topics":["generative-ai","good-first-issue","llm","llm-eval","llm-evaluation","llm-evaluation-framework","llm-evaluation-toolkit","llm-tools","llmops","llms-benchmarking","metrics","prompt-engineering"],"archived":false,"github_pushed_at":"2025-02-13T14:48:37+00:00","maintenance_label":"Dormant","stars_delta_30d":0,"url":"https://www.graphcanon.com/tools/parea-ai-parea-sdk-py","markdown_url":"https://www.graphcanon.com/tools/parea-ai-parea-sdk-py.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/parea-ai-parea-sdk-py","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=parea-ai-parea-sdk-py"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"},{"slug":"model-training","name":"Model Training","url":"https://www.graphcanon.com/categories/model-training","markdown_url":"https://www.graphcanon.com/categories/model-training.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/model-training"}],"tags":[{"slug":"llm-eval","name":"llm-eval"},{"slug":"llm-evaluation","name":"llm-evaluation"},{"slug":"prompt-engineering","name":"prompt-engineering"}],"edges":[],"neighbours":[{"slug":"promptfoo-promptfoo","name":"promptfoo","tagline":"Test prompts, agents, and RAGs. Compare performance of various LLMs.","github_url":"https://github.com/promptfoo/promptfoo","owner":"promptfoo","repo":"promptfoo","owner_avatar_url":"https://avatars.githubusercontent.com/u/137907881?v=4","primary_language":"TypeScript","stars":25251,"forks":2334,"topics":["ci","ci-cd","cicd","evaluation","evaluation-framework","llm","llm-eval","llm-evaluation","llm-evaluation-framework","llmops","pentesting","prompt-engineering","prompt-testing","prompts","rag","red-teaming","testing","vulnerability-scanners"],"archived":false,"github_pushed_at":"2026-09-18T07:59:45+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/promptfoo-promptfoo","markdown_url":"https://www.graphcanon.com/tools/promptfoo-promptfoo.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/promptfoo-promptfoo","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=promptfoo-promptfoo","shared_categories":["evaluation-observability"]},{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":18342,"forks":1953,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-09-18T17:06:58+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"raga-ai-hub-ragaai-catalyst","name":"RagaAI-Catalyst","tagline":"Python SDK for AI agent observability and evaluation","github_url":"https://github.com/raga-ai-hub/RagaAI-Catalyst","owner":"raga-ai-hub","repo":"RagaAI-Catalyst","owner_avatar_url":"https://avatars.githubusercontent.com/u/161833182?v=4","primary_language":"Python","stars":16162,"forks":3567,"topics":["agentic-ai","agentic-ai-development","agentneo","agents","ai-agent-monitoring","ai-application-debugging","ai-evaluation-tools","ai-performance-optimization","ai-tool-interaction-monitoring","llm-testing","llm-tracing","llmops"],"archived":false,"github_pushed_at":"2026-02-11T14:43:33+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/raga-ai-hub-ragaai-catalyst","markdown_url":"https://www.graphcanon.com/tools/raga-ai-hub-ragaai-catalyst.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/raga-ai-hub-ragaai-catalyst","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=raga-ai-hub-ragaai-catalyst","shared_categories":["evaluation-observability"]},{"slug":"agentops-ai-agentops","name":"agentops","tagline":"Python SDK for AI agent monitoring and LLM cost tracking","github_url":"https://github.com/AgentOps-AI/agentops","owner":"AgentOps-AI","repo":"agentops","owner_avatar_url":"https://avatars.githubusercontent.com/u/140554352?v=4","primary_language":"Python","stars":5830,"forks":625,"topics":["agent","agentops","agents-sdk","ai","anthropic","autogen","cost-estimation","crewai","evals","evaluation-metrics","groq","langchain","llm","mistral","ollama","openai","openai-agents"],"archived":false,"github_pushed_at":"2026-06-25T08:25:03+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/agentops-ai-agentops","markdown_url":"https://www.graphcanon.com/tools/agentops-ai-agentops.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/agentops-ai-agentops","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=agentops-ai-agentops","shared_categories":["evaluation-observability"]},{"slug":"pezzolabs-pezzo","name":"pezzo","tagline":"Open-source, developer-first LLMOps platform","github_url":"https://github.com/pezzolabs/pezzo","owner":"pezzolabs","repo":"pezzo","owner_avatar_url":"https://avatars.githubusercontent.com/u/119122072?v=4","primary_language":"TypeScript","stars":3273,"forks":279,"topics":["ai","devtools","gpt-3","gpt-4","hacktoberfest","javascript","langchain","llm","llmops","monitoring","nestjs","nodejs","observability","openai","platform","prompt","prompt-engineering","prompt-management","python","typescript"],"archived":false,"github_pushed_at":"2026-08-21T21:40:31+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/pezzolabs-pezzo","markdown_url":"https://www.graphcanon.com/tools/pezzolabs-pezzo.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pezzolabs-pezzo","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pezzolabs-pezzo","shared_categories":["evaluation-observability"]},{"slug":"hegelai-prompttools","name":"prompttools","tagline":"Open-source tools for prompt testing and experimentation","github_url":"https://github.com/hegelai/prompttools","owner":"hegelai","repo":"prompttools","owner_avatar_url":"https://avatars.githubusercontent.com/u/136523567?v=4","primary_language":"Python","stars":3055,"forks":256,"topics":["deep-learning","developer-tools","embeddings","large-language-models","llms","machine-learning","prompt-engineering","python","vector-search"],"archived":false,"github_pushed_at":"2026-02-11T03:24:04+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/hegelai-prompttools","markdown_url":"https://www.graphcanon.com/tools/hegelai-prompttools.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/hegelai-prompttools","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=hegelai-prompttools","shared_categories":[]},{"slug":"future-agi-future-agi","name":"future-agi","tagline":"Open-source, end-to-end platform for evaluating, observing, and improving LLM and AI agent applications","github_url":"https://github.com/future-agi/future-agi","owner":"future-agi","repo":"future-agi","owner_avatar_url":"https://avatars.githubusercontent.com/u/147392366?v=4","primary_language":"Python","stars":2032,"forks":627,"topics":["ai-agents","ai-evals","ai-gateway","ai-optimization","ai-simulations","evaluation-framework","guardrails","hallucination-detection","llm","llm-evaluation","llm-observability","llmops","model-evaluation","observability","opentelemetry","rag","rag-evaluation","simulation","telemetry","tracing"],"archived":false,"github_pushed_at":"2026-09-18T08:15:25+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/future-agi-future-agi","markdown_url":"https://www.graphcanon.com/tools/future-agi-future-agi.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/future-agi-future-agi","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=future-agi-future-agi","shared_categories":["evaluation-observability"]},{"slug":"e2b-dev-awesome-ai-sdks","name":"awesome-ai-sdks","tagline":"A database of SDKs for AI agents creation and management","github_url":"https://github.com/e2b-dev/awesome-ai-sdks","owner":"e2b-dev","repo":"awesome-ai-sdks","owner_avatar_url":"https://avatars.githubusercontent.com/u/129434473?v=4","primary_language":null,"stars":1223,"forks":399,"topics":["agent","agentops","agents","ai","ai-agents","awesome","awesome-list","chatgpt","e2b","framework","langchain","llama-index","llm","llmops","openai","sdk","tools","vercel"],"archived":false,"github_pushed_at":"2026-07-09T17:29:58+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/e2b-dev-awesome-ai-sdks","markdown_url":"https://www.graphcanon.com/tools/e2b-dev-awesome-ai-sdks.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/e2b-dev-awesome-ai-sdks","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=e2b-dev-awesome-ai-sdks","shared_categories":[]},{"slug":"declare-lab-instruct-eval","name":"instruct-eval","tagline":"Quantitative evaluation for instruction-tuned language models","github_url":"https://github.com/declare-lab/instruct-eval","owner":"declare-lab","repo":"instruct-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/59164695?v=4","primary_language":"Python","stars":553,"forks":45,"topics":["instruct-tuning","llm"],"archived":false,"github_pushed_at":"2024-03-10T05:00:00+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval","markdown_url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/declare-lab-instruct-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=declare-lab-instruct-eval","shared_categories":["evaluation-observability"]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]},{"slug":"emirsahin1-llm-axe","name":"llm-axe","tagline":"Toolkit for quick implementation of LLM powered applications","github_url":"https://github.com/emirsahin1/llm-axe","owner":"emirsahin1","repo":"llm-axe","owner_avatar_url":"https://avatars.githubusercontent.com/u/50391065?v=4","primary_language":"Python","stars":275,"forks":38,"topics":["function-calling","llama3","llm","local-llm","ollama","pdf-llm"],"archived":false,"github_pushed_at":"2025-01-05T19:47:01+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/emirsahin1-llm-axe","markdown_url":"https://www.graphcanon.com/tools/emirsahin1-llm-axe.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/emirsahin1-llm-axe","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=emirsahin1-llm-axe","shared_categories":["model-training"]},{"slug":"hidai25-eval-view","name":"eval-view","tagline":"Regression testing for AI agents, snapshots behavior, diffs tool calls, catches regressions in CI","github_url":"https://github.com/hidai25/eval-view","owner":"hidai25","repo":"eval-view","owner_avatar_url":"https://avatars.githubusercontent.com/u/31502796?v=4","primary_language":"Python","stars":134,"forks":24,"topics":["agent-benchmark","agent-evaluation","agentic-ai","ai-agents","anthropic","autogen","cli","crewai","evaluation","langchain-agent","langgraph","llm","mcp","openai-assistants","pytest","python","regression-testing","testing"],"archived":false,"github_pushed_at":"2026-09-05T03:28:43+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/hidai25-eval-view","markdown_url":"https://www.graphcanon.com/tools/hidai25-eval-view.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/hidai25-eval-view","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=hidai25-eval-view","shared_categories":["evaluation-observability"]}]}}