{"data":{"node":{"slug":"relari-ai-continuous-eval","name":"continuous-eval","tagline":"Data-Driven Evaluation for LLM-Powered Applications","github_url":"https://github.com/relari-ai/continuous-eval","owner":"relari-ai","repo":"continuous-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/135984758?v=4","primary_language":"Python","stars":515,"forks":38,"topics":["evaluation-framework","evaluation-metrics","information-retrieval","llm-evaluation","llmops","rag","retrieval-augmented-generation"],"archived":false,"github_pushed_at":"2026-08-10T22:12:03+00:00","maintenance_label":"Active","stars_delta_30d":-1,"url":"https://www.graphcanon.com/tools/relari-ai-continuous-eval","markdown_url":"https://www.graphcanon.com/tools/relari-ai-continuous-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/relari-ai-continuous-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=relari-ai-continuous-eval"},"categories":[{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"},{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"evaluation-framework","name":"evaluation-framework"},{"slug":"evaluation-metrics","name":"evaluation-metrics"},{"slug":"information-retrieval","name":"information-retrieval"},{"slug":"llm-evaluation","name":"llm-evaluation"},{"slug":"llmops","name":"llmops"},{"slug":"rag","name":"rag"},{"slug":"retrieval-augmented-generation","name":"retrieval-augmented-generation"}],"edges":[{"type":"alternative","direction":"out","explanation":"`continuous-eval` and `RagaAI-Catalyst` both offer frameworks for monitoring, evaluating LLM applications.","successor_context":null,"tool":{"slug":"raga-ai-hub-ragaai-catalyst","name":"RagaAI-Catalyst","tagline":"Python SDK for AI agent observability and evaluation","github_url":"https://github.com/raga-ai-hub/RagaAI-Catalyst","owner":"raga-ai-hub","repo":"RagaAI-Catalyst","owner_avatar_url":"https://avatars.githubusercontent.com/u/161833182?v=4","primary_language":"Python","stars":16148,"forks":3565,"topics":["agentic-ai","agentic-ai-development","agentneo","agents","ai-agent-monitoring","ai-application-debugging","ai-evaluation-tools","ai-performance-optimization","ai-tool-interaction-monitoring","llm-testing","llm-tracing","llmops"],"archived":false,"github_pushed_at":"2026-02-11T14:43:33+00:00","maintenance_label":"Slowing","stars_delta_30d":5,"url":"https://www.graphcanon.com/tools/raga-ai-hub-ragaai-catalyst","markdown_url":"https://www.graphcanon.com/tools/raga-ai-hub-ragaai-catalyst.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/raga-ai-hub-ragaai-catalyst","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=raga-ai-hub-ragaai-catalyst"}},{"type":"alternative","direction":"out","explanation":"`continuous-eval` and `langfuse-langfuse` both provide evaluation frameworks for LLMs, albeit with different focuses.","successor_context":null,"tool":{"slug":"langfuse-langfuse","name":"langfuse","tagline":"Open source AI engineering platform: LLM evals, observability, metrics, prompt management, playground, datasets","github_url":"https://github.com/langfuse/langfuse","owner":"langfuse","repo":"langfuse","owner_avatar_url":"https://avatars.githubusercontent.com/u/134601687?v=4","primary_language":"TypeScript","stars":32271,"forks":3466,"topics":["analytics","autogen","evaluation","langchain","large-language-models","llama-index","llm","llm-evaluation","llm-observability","llmops","monitoring","observability","open-source","openai","playground","prompt-engineering","prompt-management","self-hosted","ycombinator"],"archived":false,"github_pushed_at":"2026-07-31T22:58:07+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/langfuse-langfuse","markdown_url":"https://www.graphcanon.com/tools/langfuse-langfuse.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langfuse-langfuse","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langfuse-langfuse"}},{"type":"alternative","direction":"out","explanation":"`continuous-eval` and `Evidently` both serve as observability frameworks for ML and LLM systems, emphasizing evaluation aspects.","successor_context":null,"tool":{"slug":"evidentlyai-evidently","name":"evidently","tagline":"An open-source ML and LLM observability framework.","github_url":"https://github.com/evidentlyai/evidently","owner":"evidentlyai","repo":"evidently","owner_avatar_url":"https://avatars.githubusercontent.com/u/75031056?v=4","primary_language":"Jupyter Notebook","stars":7790,"forks":895,"topics":["data-drift","data-quality","data-science","data-validation","generative-ai","hacktoberfest","html-report","jupyter-notebook","llm","llmops","machine-learning","mlops","model-monitoring","pandas-dataframe"],"archived":false,"github_pushed_at":"2026-08-05T16:29:57+00:00","maintenance_label":"Very active","stars_delta_30d":117,"url":"https://www.graphcanon.com/tools/evidentlyai-evidently","markdown_url":"https://www.graphcanon.com/tools/evidentlyai-evidently.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evidentlyai-evidently","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evidentlyai-evidently"}},{"type":"related","direction":"out","explanation":null,"successor_context":null,"tool":{"slug":"langwatch-langwatch","name":"langwatch","tagline":"The platform for LLM evaluations and AI agent testing","github_url":"https://github.com/langwatch/langwatch","owner":"langwatch","repo":"langwatch","owner_avatar_url":"https://avatars.githubusercontent.com/u/146763322?v=4","primary_language":"TypeScript","stars":3479,"forks":340,"topics":["ai","analytics","datasets","dspy","evaluation","gpt","llm","llm-ops","llmops","low-code","observability","openai","prompt-engineering"],"archived":false,"github_pushed_at":"2026-08-07T21:03:52+00:00","maintenance_label":"Very active","stars_delta_30d":152,"url":"https://www.graphcanon.com/tools/langwatch-langwatch","markdown_url":"https://www.graphcanon.com/tools/langwatch-langwatch.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langwatch-langwatch","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langwatch-langwatch"}},{"type":"related","direction":"out","explanation":null,"successor_context":null,"tool":{"slug":"agenta-ai-agenta","name":"agenta","tagline":"The open-source LLMOps platform for prompt management, evaluation, and observability.","github_url":"https://github.com/Agenta-AI/agenta","owner":"Agenta-AI","repo":"agenta","owner_avatar_url":"https://avatars.githubusercontent.com/u/127993667?v=4","primary_language":"TypeScript","stars":4445,"forks":609,"topics":["agent-builder","agent-observability","agent-orchestration","agent-workspace","agentic-ai","ai-agent","ai-agents","ai-automation","ai-skills-manager","ai-workflow-builder","harness","mcp","open-source","self-hosted","workflow-automation"],"archived":false,"github_pushed_at":"2026-08-07T10:41:36+00:00","maintenance_label":"Very active","stars_delta_30d":170,"url":"https://www.graphcanon.com/tools/agenta-ai-agenta","markdown_url":"https://www.graphcanon.com/tools/agenta-ai-agenta.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/agenta-ai-agenta","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=agenta-ai-agenta"}},{"type":"alternative","direction":"out","explanation":"Both `continuous-eval` and `ragas` aim to provide comprehensive evaluation capabilities for LLM applications, making them alternatives.","successor_context":null,"tool":{"slug":"vibrantlabsai-ragas","name":"ragas","tagline":"Supercharge Your LLM Application Evaluations 🚀","github_url":"https://github.com/vibrantlabsai/ragas","owner":"vibrantlabsai","repo":"ragas","owner_avatar_url":"https://avatars.githubusercontent.com/u/122604797?v=4","primary_language":"Python","stars":15388,"forks":1637,"topics":["evaluation","llm","llmops"],"archived":false,"github_pushed_at":"2026-02-24T07:47:19+00:00","maintenance_label":"Slowing","stars_delta_30d":470,"url":"https://www.graphcanon.com/tools/vibrantlabsai-ragas","markdown_url":"https://www.graphcanon.com/tools/vibrantlabsai-ragas.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vibrantlabsai-ragas","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vibrantlabsai-ragas"}}],"neighbours":[{"slug":"langfuse-langfuse","name":"langfuse","tagline":"Open source AI engineering platform: LLM evals, observability, metrics, prompt management, playground, datasets","github_url":"https://github.com/langfuse/langfuse","owner":"langfuse","repo":"langfuse","owner_avatar_url":"https://avatars.githubusercontent.com/u/134601687?v=4","primary_language":"TypeScript","stars":32271,"forks":3466,"topics":["analytics","autogen","evaluation","langchain","large-language-models","llama-index","llm","llm-evaluation","llm-observability","llmops","monitoring","observability","open-source","openai","playground","prompt-engineering","prompt-management","self-hosted","ycombinator"],"archived":false,"github_pushed_at":"2026-07-31T22:58:07+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/langfuse-langfuse","markdown_url":"https://www.graphcanon.com/tools/langfuse-langfuse.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langfuse-langfuse","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langfuse-langfuse","shared_categories":["evaluation-observability"]},{"slug":"openai-evals","name":"evals","tagline":"Framework for evaluating LLMs and LLM systems with an open-source registry of benchmarks.","github_url":"https://github.com/openai/evals","owner":"openai","repo":"evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/14957082?v=4","primary_language":"Python","stars":19127,"forks":3050,"topics":[],"archived":false,"github_pushed_at":"2026-04-14T15:29:57+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/openai-evals","markdown_url":"https://www.graphcanon.com/tools/openai-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openai-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openai-evals","shared_categories":["evaluation-observability"]},{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":17226,"forks":1736,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-07-27T11:33:31+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"vibrantlabsai-ragas","name":"ragas","tagline":"Supercharge Your LLM Application Evaluations 🚀","github_url":"https://github.com/vibrantlabsai/ragas","owner":"vibrantlabsai","repo":"ragas","owner_avatar_url":"https://avatars.githubusercontent.com/u/122604797?v=4","primary_language":"Python","stars":15388,"forks":1637,"topics":["evaluation","llm","llmops"],"archived":false,"github_pushed_at":"2026-02-24T07:47:19+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/vibrantlabsai-ragas","markdown_url":"https://www.graphcanon.com/tools/vibrantlabsai-ragas.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vibrantlabsai-ragas","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vibrantlabsai-ragas","shared_categories":["evaluation-observability"]},{"slug":"open-compass-vlmevalkit","name":"VLMEvalKit","tagline":"An open-source evaluation toolkit for large vision-language models","github_url":"https://github.com/open-compass/VLMEvalKit","owner":"open-compass","repo":"VLMEvalKit","owner_avatar_url":"https://avatars.githubusercontent.com/u/143521324?v=4","primary_language":"Python","stars":4345,"forks":745,"topics":["chatgpt","claude","clip","computer-vision","evaluation","gemini","gpt","gpt-4v","gpt4","large-language-models","llava","llm","multi-modal","openai","openai-api","pytorch","qwen","vit","vqa"],"archived":false,"github_pushed_at":"2026-08-17T17:16:26+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/open-compass-vlmevalkit","markdown_url":"https://www.graphcanon.com/tools/open-compass-vlmevalkit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/open-compass-vlmevalkit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=open-compass-vlmevalkit","shared_categories":["evaluation-observability"]},{"slug":"truera-trulens","name":"trulens","tagline":"Evaluation and Tracking for LLM Experiments and AI Agents","github_url":"https://github.com/truera/trulens","owner":"truera","repo":"trulens","owner_avatar_url":"https://avatars.githubusercontent.com/u/51224128?v=4","primary_language":"Python","stars":3516,"forks":327,"topics":["agent-evaluation","agentops","ai-agents","ai-monitoring","ai-observability","evals","explainable-ml","llm-eval","llm-evaluation","llmops","llms","machine-learning","neural-networks"],"archived":false,"github_pushed_at":"2026-08-20T10:21:00+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/truera-trulens","markdown_url":"https://www.graphcanon.com/tools/truera-trulens.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/truera-trulens","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=truera-trulens","shared_categories":["evaluation-observability"]},{"slug":"stanford-crfm-helm","name":"helm","tagline":"Holistic, reproducible and transparent evaluation of foundation models","github_url":"https://github.com/stanford-crfm/helm","owner":"stanford-crfm","repo":"helm","owner_avatar_url":"https://avatars.githubusercontent.com/u/75054807?v=4","primary_language":"Python","stars":2873,"forks":406,"topics":[],"archived":false,"github_pushed_at":"2026-08-01T01:23:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/stanford-crfm-helm","markdown_url":"https://www.graphcanon.com/tools/stanford-crfm-helm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/stanford-crfm-helm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=stanford-crfm-helm","shared_categories":["evaluation-observability"]},{"slug":"huggingface-lighteval","name":"lighteval","tagline":"All-in-one toolkit for evaluating LLMs across multiple backends","github_url":"https://github.com/huggingface/lighteval","owner":"huggingface","repo":"lighteval","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":2508,"forks":523,"topics":["evaluation","evaluation-framework","evaluation-metrics","huggingface"],"archived":false,"github_pushed_at":"2026-06-29T13:03:33+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huggingface-lighteval","markdown_url":"https://www.graphcanon.com/tools/huggingface-lighteval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-lighteval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-lighteval","shared_categories":["evaluation-observability"]},{"slug":"evalplus-evalplus","name":"evalplus","tagline":"Rigorous evaluation of LLM-synthesized code","github_url":"https://github.com/evalplus/evalplus","owner":"evalplus","repo":"evalplus","owner_avatar_url":"https://avatars.githubusercontent.com/u/132106461?v=4","primary_language":"Python","stars":1794,"forks":205,"topics":["benchmark","chatgpt","efficiency","gpt-4","large-language-models","program-synthesis","testing"],"archived":false,"github_pushed_at":"2025-10-02T22:56:38+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/evalplus-evalplus","markdown_url":"https://www.graphcanon.com/tools/evalplus-evalplus.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evalplus-evalplus","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evalplus-evalplus","shared_categories":["evaluation-observability"]},{"slug":"prometheus-eval-prometheus-eval","name":"prometheus-eval","tagline":"Evaluate your LLM's response with Prometheus and GPT4","github_url":"https://github.com/prometheus-eval/prometheus-eval","owner":"prometheus-eval","repo":"prometheus-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/167460660?v=4","primary_language":"Python","stars":1107,"forks":68,"topics":["evaluation","gpt4","litellm","llm","llm-as-a-judge","llm-as-evaluator","llmops","python","vllm"],"archived":false,"github_pushed_at":"2025-04-25T03:58:37+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/prometheus-eval-prometheus-eval","markdown_url":"https://www.graphcanon.com/tools/prometheus-eval-prometheus-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/prometheus-eval-prometheus-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=prometheus-eval-prometheus-eval","shared_categories":["evaluation-observability"]},{"slug":"rlancemartin-auto-evaluator","name":"auto-evaluator","tagline":"A lightweight evaluation tool for question-answering using Langchain","github_url":"https://github.com/rlancemartin/auto-evaluator","owner":"rlancemartin","repo":"auto-evaluator","owner_avatar_url":"https://avatars.githubusercontent.com/u/122662504?v=4","primary_language":"Python","stars":1105,"forks":92,"topics":[],"archived":false,"github_pushed_at":"2023-05-10T02:00:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator","markdown_url":"https://www.graphcanon.com/tools/rlancemartin-auto-evaluator.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rlancemartin-auto-evaluator","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rlancemartin-auto-evaluator","shared_categories":["evaluation-observability"]},{"slug":"judgmentlabs-judgeval","name":"judgeval","tagline":"The Continuous-Improvement Stack for Agents","github_url":"https://github.com/JudgmentLabs/judgeval","owner":"JudgmentLabs","repo":"judgeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/186249420?v=4","primary_language":"Python","stars":1047,"forks":95,"topics":["agent","agentic-ai","agents","grpo","langchain","langgraph","llama-index","llm","llm-evaluation","llm-observability","open-source","openai","prompt-engineering","reinforcement-learning","rl"],"archived":false,"github_pushed_at":"2026-07-27T00:46:56+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/judgmentlabs-judgeval","markdown_url":"https://www.graphcanon.com/tools/judgmentlabs-judgeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/judgmentlabs-judgeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=judgmentlabs-judgeval","shared_categories":["evaluation-observability"]}]}}