{"data":{"node":{"slug":"truera-trulens","name":"trulens","tagline":"Evaluation and Tracking for LLM Experiments and AI Agents","github_url":"https://github.com/truera/trulens","owner":"truera","repo":"trulens","owner_avatar_url":"https://avatars.githubusercontent.com/u/51224128?v=4","primary_language":"Python","stars":3516,"forks":327,"topics":["agent-evaluation","agentops","ai-agents","ai-monitoring","ai-observability","evals","explainable-ml","llm-eval","llm-evaluation","llmops","llms","machine-learning","neural-networks"],"archived":false,"github_pushed_at":"2026-08-20T10:21:00+00:00","maintenance_label":"Very active","stars_delta_30d":68,"url":"https://www.graphcanon.com/tools/truera-trulens","markdown_url":"https://www.graphcanon.com/tools/truera-trulens.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/truera-trulens","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=truera-trulens"},"categories":[{"slug":"ai-agents","name":"AI Agents","url":"https://www.graphcanon.com/categories/ai-agents","markdown_url":"https://www.graphcanon.com/categories/ai-agents.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/ai-agents"},{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"agent-evaluation","name":"agent-evaluation"},{"slug":"ai-agents","name":"ai-agents"},{"slug":"ai-monitoring","name":"ai-monitoring"},{"slug":"evaluation-tool","name":"evaluation-tool"},{"slug":"llm-eval","name":"llm-eval"},{"slug":"observability","name":"observability"}],"edges":[{"type":"alternative","direction":"out","explanation":"RagaAI-Catalyst offers observability for agents, much like TruLens, but focuses more on an SDK while TruLens centers around systematic evaluation and feedback.","successor_context":null,"tool":{"slug":"raga-ai-hub-ragaai-catalyst","name":"RagaAI-Catalyst","tagline":"Python SDK for AI agent observability and evaluation","github_url":"https://github.com/raga-ai-hub/RagaAI-Catalyst","owner":"raga-ai-hub","repo":"RagaAI-Catalyst","owner_avatar_url":"https://avatars.githubusercontent.com/u/161833182?v=4","primary_language":"Python","stars":16148,"forks":3565,"topics":["agentic-ai","agentic-ai-development","agentneo","agents","ai-agent-monitoring","ai-application-debugging","ai-evaluation-tools","ai-performance-optimization","ai-tool-interaction-monitoring","llm-testing","llm-tracing","llmops"],"archived":false,"github_pushed_at":"2026-02-11T14:43:33+00:00","maintenance_label":"Slowing","stars_delta_30d":5,"url":"https://www.graphcanon.com/tools/raga-ai-hub-ragaai-catalyst","markdown_url":"https://www.graphcanon.com/tools/raga-ai-hub-ragaai-catalyst.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/raga-ai-hub-ragaai-catalyst","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=raga-ai-hub-ragaai-catalyst"}},{"type":"alternative","direction":"out","explanation":"LangFuse and TruLens both focus on LLM evals, observability, metrics, and offer tools to evaluate and improve applications systematically.","successor_context":null,"tool":{"slug":"langfuse-langfuse","name":"langfuse","tagline":"Open source AI engineering platform: LLM evals, observability, metrics, prompt management, playground, datasets","github_url":"https://github.com/langfuse/langfuse","owner":"langfuse","repo":"langfuse","owner_avatar_url":"https://avatars.githubusercontent.com/u/134601687?v=4","primary_language":"TypeScript","stars":32271,"forks":3466,"topics":["analytics","autogen","evaluation","langchain","large-language-models","llama-index","llm","llm-evaluation","llm-observability","llmops","monitoring","observability","open-source","openai","playground","prompt-engineering","prompt-management","self-hosted","ycombinator"],"archived":false,"github_pushed_at":"2026-07-31T22:58:07+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/langfuse-langfuse","markdown_url":"https://www.graphcanon.com/tools/langfuse-langfuse.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langfuse-langfuse","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langfuse-langfuse"}},{"type":"alternative","direction":"out","explanation":"Both LANGTRACE and TruLens provide open-source observability solutions for LLMs, helping with monitoring and evaluation during development phases.","successor_context":null,"tool":{"slug":"scale3-labs-langtrace","name":"langtrace","tagline":"Open Telemetry based observability tool for LLM applications","github_url":"https://github.com/Scale3-Labs/langtrace","owner":"Scale3-Labs","repo":"langtrace","owner_avatar_url":"https://avatars.githubusercontent.com/u/110545750?v=4","primary_language":"TypeScript","stars":1228,"forks":126,"topics":["ai","datasets","evaluations","gpt","langchain","llm","llm-framework","llmops","observability","open-source","open-telemetry","openai","prompt-engineering","tracing"],"archived":false,"github_pushed_at":"2025-11-17T15:08:48+00:00","maintenance_label":"Slowing","stars_delta_30d":12,"url":"https://www.graphcanon.com/tools/scale3-labs-langtrace","markdown_url":"https://www.graphcanon.com/tools/scale3-labs-langtrace.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/scale3-labs-langtrace","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=scale3-labs-langtrace"}},{"type":"alternative","direction":"out","explanation":"Evidently is another observability framework that deals with ML and LLMs similar to TruLens, allowing for detailed tracking and evaluation during the development cycle.","successor_context":null,"tool":{"slug":"evidentlyai-evidently","name":"evidently","tagline":"An open-source ML and LLM observability framework.","github_url":"https://github.com/evidentlyai/evidently","owner":"evidentlyai","repo":"evidently","owner_avatar_url":"https://avatars.githubusercontent.com/u/75031056?v=4","primary_language":"Jupyter Notebook","stars":7790,"forks":895,"topics":["data-drift","data-quality","data-science","data-validation","generative-ai","hacktoberfest","html-report","jupyter-notebook","llm","llmops","machine-learning","mlops","model-monitoring","pandas-dataframe"],"archived":false,"github_pushed_at":"2026-08-05T16:29:57+00:00","maintenance_label":"Very active","stars_delta_30d":117,"url":"https://www.graphcanon.com/tools/evidentlyai-evidently","markdown_url":"https://www.graphcanon.com/tools/evidentlyai-evidently.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evidentlyai-evidently","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evidentlyai-evidently"}},{"type":"alternative","direction":"out","explanation":"Trulens and LangWatch both offer tools for evaluating and monitoring Large Language Models (LLMs) and AI agents, with Trulens focusing on systematic evaluation and tracking through fine-grained instrumentation, while LangWatch emphasizes regression testing, simulation, and production observability. Their overlapping functionalities in LLM evaluation establish an 'alternative' relationship between","successor_context":null,"tool":{"slug":"langwatch-langwatch","name":"langwatch","tagline":"The platform for LLM evaluations and AI agent testing","github_url":"https://github.com/langwatch/langwatch","owner":"langwatch","repo":"langwatch","owner_avatar_url":"https://avatars.githubusercontent.com/u/146763322?v=4","primary_language":"TypeScript","stars":3479,"forks":340,"topics":["ai","analytics","datasets","dspy","evaluation","gpt","llm","llm-ops","llmops","low-code","observability","openai","prompt-engineering"],"archived":false,"github_pushed_at":"2026-08-07T21:03:52+00:00","maintenance_label":"Very active","stars_delta_30d":152,"url":"https://www.graphcanon.com/tools/langwatch-langwatch","markdown_url":"https://www.graphcanon.com/tools/langwatch-langwatch.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langwatch-langwatch","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langwatch-langwatch"}},{"type":"alternative","direction":"out","explanation":"Both Comet.ml and TruLens offer AI observability, evaluation, and optimization solutions, providing feedback on model performance during development.","successor_context":null,"tool":{"slug":"comet-ml-opik","name":"opik","tagline":"Debug, evaluate, and monitor your LLM applications with comprehensive tracing and production-ready dashboards","github_url":"https://github.com/comet-ml/opik","owner":"comet-ml","repo":"opik","owner_avatar_url":"https://avatars.githubusercontent.com/u/31487821?v=4","primary_language":"Python","stars":21177,"forks":1681,"topics":["evaluation","hacktoberfest","hacktoberfest2025","langchain","llama-index","llm","llm-evaluation","llm-observability","llmops","open-source","openai","playground","prompt-engineering"],"archived":false,"github_pushed_at":"2026-08-07T11:46:34+00:00","maintenance_label":"Very active","stars_delta_30d":767,"url":"https://www.graphcanon.com/tools/comet-ml-opik","markdown_url":"https://www.graphcanon.com/tools/comet-ml-opik.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/comet-ml-opik","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=comet-ml-opik"}},{"type":"related","direction":"out","explanation":null,"successor_context":null,"tool":{"slug":"agenta-ai-agenta","name":"agenta","tagline":"The open-source LLMOps platform for prompt management, evaluation, and observability.","github_url":"https://github.com/Agenta-AI/agenta","owner":"Agenta-AI","repo":"agenta","owner_avatar_url":"https://avatars.githubusercontent.com/u/127993667?v=4","primary_language":"TypeScript","stars":4445,"forks":609,"topics":["agent-builder","agent-observability","agent-orchestration","agent-workspace","agentic-ai","ai-agent","ai-agents","ai-automation","ai-skills-manager","ai-workflow-builder","harness","mcp","open-source","self-hosted","workflow-automation"],"archived":false,"github_pushed_at":"2026-08-07T10:41:36+00:00","maintenance_label":"Very active","stars_delta_30d":170,"url":"https://www.graphcanon.com/tools/agenta-ai-agenta","markdown_url":"https://www.graphcanon.com/tools/agenta-ai-agenta.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/agenta-ai-agenta","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=agenta-ai-agenta"}},{"type":"alternative","direction":"out","explanation":"OpenLit shares a similar focus on AI engineering observability but is more focused on providing a platform with diverse functionalities whereas TruLens specializes in detailed evaluation and monitoring.","successor_context":null,"tool":{"slug":"openlit-openlit","name":"openlit","tagline":"A comprehensive open-source platform for AI Engineering with LLM Observability, Monitoring, and Management","github_url":"https://github.com/openlit/openlit","owner":"openlit","repo":"openlit","owner_avatar_url":"https://avatars.githubusercontent.com/u/149867240?v=4","primary_language":"TypeScript","stars":2664,"forks":342,"topics":["ai-observability","amd-gpu","clickhouse","distributed-tracing","genai","gpu-monitoring","grafana","langchain","llmops","llms","metrics","monitoring-tool","nvidia-smi","observability","open-source","openai","opentelemetry","otlp","python","tracing"],"archived":false,"github_pushed_at":"2026-07-31T18:39:37+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/openlit-openlit","markdown_url":"https://www.graphcanon.com/tools/openlit-openlit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openlit-openlit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openlit-openlit"}},{"type":"alternative","direction":"out","explanation":"OpenLLMetry provides open-source observability for LLMs, similar to TruLens, which focuses on evaluation and tracking of performance across developmental iterations.","successor_context":null,"tool":{"slug":"traceloop-openllmetry","name":"openllmetry","tagline":"Open-source observability for GenAI and LLM applications based on OpenTelemetry.","github_url":"https://github.com/traceloop/openllmetry","owner":"traceloop","repo":"openllmetry","owner_avatar_url":"https://avatars.githubusercontent.com/u/125419530?v=4","primary_language":"Python","stars":7377,"forks":1047,"topics":["artifical-intelligence","datascience","generative-ai","good-first-issue","good-first-issues","help-wanted","llm","llmops","metrics","ml","model-monitoring","monitoring","observability","open-source","open-telemetry","opentelemetry","opentelemetry-python","python"],"archived":false,"github_pushed_at":"2026-08-10T08:49:01+00:00","maintenance_label":"Very active","stars_delta_30d":75,"url":"https://www.graphcanon.com/tools/traceloop-openllmetry","markdown_url":"https://www.graphcanon.com/tools/traceloop-openllmetry.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/traceloop-openllmetry","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=traceloop-openllmetry"}},{"type":"related","direction":"in","explanation":null,"successor_context":null,"tool":{"slug":"vectordotdev-vector","name":"vector","tagline":"A high-performance observability data pipeline","github_url":"https://github.com/vectordotdev/vector","owner":"vectordotdev","repo":"vector","owner_avatar_url":"https://avatars.githubusercontent.com/u/16866914?v=4","primary_language":"Rust","stars":22396,"forks":2258,"topics":["agent","cloud-native","data-transformation","datadog","etl","events","forwarder","hacktoberfest","high-performance","logs","metrics","monitoring","observability","pipelines","rust-lang","stream-processing","telemetry","traces"],"archived":false,"github_pushed_at":"2026-08-18T21:55:24+00:00","maintenance_label":"Very active","stars_delta_30d":198,"url":"https://www.graphcanon.com/tools/vectordotdev-vector","markdown_url":"https://www.graphcanon.com/tools/vectordotdev-vector.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vectordotdev-vector","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vectordotdev-vector"}},{"type":"related","direction":"in","explanation":null,"successor_context":null,"tool":{"slug":"vibrantlabsai-ragas","name":"ragas","tagline":"Supercharge Your LLM Application Evaluations 🚀","github_url":"https://github.com/vibrantlabsai/ragas","owner":"vibrantlabsai","repo":"ragas","owner_avatar_url":"https://avatars.githubusercontent.com/u/122604797?v=4","primary_language":"Python","stars":15388,"forks":1637,"topics":["evaluation","llm","llmops"],"archived":false,"github_pushed_at":"2026-02-24T07:47:19+00:00","maintenance_label":"Slowing","stars_delta_30d":470,"url":"https://www.graphcanon.com/tools/vibrantlabsai-ragas","markdown_url":"https://www.graphcanon.com/tools/vibrantlabsai-ragas.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/vibrantlabsai-ragas","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=vibrantlabsai-ragas"}},{"type":"alternative","direction":"in","explanation":"Both 'opik' and 'trulens' offer observability and evaluation features for AI applications, particularly focusing on LLMs. While 'opik' emphasizes open-source tools for tracing, evaluating, and optimizing generative AI applications from simple chatbots to complex agentic systems, 'trulens' is specialized in systematic evaluation and tracking of LLM experiments with a focus on identifying failure-mr","successor_context":null,"tool":{"slug":"comet-ml-opik","name":"opik","tagline":"Debug, evaluate, and monitor your LLM applications with comprehensive tracing and production-ready dashboards","github_url":"https://github.com/comet-ml/opik","owner":"comet-ml","repo":"opik","owner_avatar_url":"https://avatars.githubusercontent.com/u/31487821?v=4","primary_language":"Python","stars":21177,"forks":1681,"topics":["evaluation","hacktoberfest","hacktoberfest2025","langchain","llama-index","llm","llm-evaluation","llm-observability","llmops","open-source","openai","playground","prompt-engineering"],"archived":false,"github_pushed_at":"2026-08-07T11:46:34+00:00","maintenance_label":"Very active","stars_delta_30d":767,"url":"https://www.graphcanon.com/tools/comet-ml-opik","markdown_url":"https://www.graphcanon.com/tools/comet-ml-opik.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/comet-ml-opik","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=comet-ml-opik"}},{"type":"alternative","direction":"in","explanation":"Evidently and Trulens both offer evaluation frameworks for monitoring AI systems, with Evidently focusing on a broad range of AI observability needs across different data types using over 100 metrics, whereas Trulens specifically targets the systematic evaluation and tracking of LLM experiments by offering fine-grained instrumentation to identify failure modes. This alternative relationship stems从","successor_context":null,"tool":{"slug":"evidentlyai-evidently","name":"evidently","tagline":"An open-source ML and LLM observability framework.","github_url":"https://github.com/evidentlyai/evidently","owner":"evidentlyai","repo":"evidently","owner_avatar_url":"https://avatars.githubusercontent.com/u/75031056?v=4","primary_language":"Jupyter Notebook","stars":7790,"forks":895,"topics":["data-drift","data-quality","data-science","data-validation","generative-ai","hacktoberfest","html-report","jupyter-notebook","llm","llmops","machine-learning","mlops","model-monitoring","pandas-dataframe"],"archived":false,"github_pushed_at":"2026-08-05T16:29:57+00:00","maintenance_label":"Very active","stars_delta_30d":117,"url":"https://www.graphcanon.com/tools/evidentlyai-evidently","markdown_url":"https://www.graphcanon.com/tools/evidentlyai-evidently.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evidentlyai-evidently","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evidentlyai-evidently"}},{"type":"alternative","direction":"in","explanation":"Prometheus-eval and trulens both offer methodologies and tools for the evaluation of Large Language Models, though they approach this with differing methodologies. Prometheus-eval utilizes a specific setup involving Prometheus alongside GPT4 to conduct its evaluations, whereas TruLens offers a broader suite of tools that includes fine-grained instrumentation aimed at identifying failure modes in L","successor_context":null,"tool":{"slug":"prometheus-eval-prometheus-eval","name":"prometheus-eval","tagline":"Evaluate your LLM's response with Prometheus and GPT4","github_url":"https://github.com/prometheus-eval/prometheus-eval","owner":"prometheus-eval","repo":"prometheus-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/167460660?v=4","primary_language":"Python","stars":1107,"forks":68,"topics":["evaluation","gpt4","litellm","llm","llm-as-a-judge","llm-as-evaluator","llmops","python","vllm"],"archived":false,"github_pushed_at":"2025-04-25T03:58:37+00:00","maintenance_label":"Dormant","stars_delta_30d":5,"url":"https://www.graphcanon.com/tools/prometheus-eval-prometheus-eval","markdown_url":"https://www.graphcanon.com/tools/prometheus-eval-prometheus-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/prometheus-eval-prometheus-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=prometheus-eval-prometheus-eval"}}],"neighbours":[{"slug":"langfuse-langfuse","name":"langfuse","tagline":"Open source AI engineering platform: LLM evals, observability, metrics, prompt management, playground, datasets","github_url":"https://github.com/langfuse/langfuse","owner":"langfuse","repo":"langfuse","owner_avatar_url":"https://avatars.githubusercontent.com/u/134601687?v=4","primary_language":"TypeScript","stars":32271,"forks":3466,"topics":["analytics","autogen","evaluation","langchain","large-language-models","llama-index","llm","llm-evaluation","llm-observability","llmops","monitoring","observability","open-source","openai","playground","prompt-engineering","prompt-management","self-hosted","ycombinator"],"archived":false,"github_pushed_at":"2026-07-31T22:58:07+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/langfuse-langfuse","markdown_url":"https://www.graphcanon.com/tools/langfuse-langfuse.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langfuse-langfuse","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langfuse-langfuse","shared_categories":["evaluation-observability"]},{"slug":"comet-ml-opik","name":"opik","tagline":"Debug, evaluate, and monitor your LLM applications with comprehensive tracing and production-ready dashboards","github_url":"https://github.com/comet-ml/opik","owner":"comet-ml","repo":"opik","owner_avatar_url":"https://avatars.githubusercontent.com/u/31487821?v=4","primary_language":"Python","stars":21177,"forks":1681,"topics":["evaluation","hacktoberfest","hacktoberfest2025","langchain","llama-index","llm","llm-evaluation","llm-observability","llmops","open-source","openai","playground","prompt-engineering"],"archived":false,"github_pushed_at":"2026-08-07T11:46:34+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/comet-ml-opik","markdown_url":"https://www.graphcanon.com/tools/comet-ml-opik.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/comet-ml-opik","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=comet-ml-opik","shared_categories":["evaluation-observability"]},{"slug":"openai-evals","name":"evals","tagline":"Framework for evaluating LLMs and LLM systems with an open-source registry of benchmarks.","github_url":"https://github.com/openai/evals","owner":"openai","repo":"evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/14957082?v=4","primary_language":"Python","stars":19127,"forks":3050,"topics":[],"archived":false,"github_pushed_at":"2026-04-14T15:29:57+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/openai-evals","markdown_url":"https://www.graphcanon.com/tools/openai-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openai-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openai-evals","shared_categories":["evaluation-observability"]},{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":17226,"forks":1736,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-07-27T11:33:31+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"raga-ai-hub-ragaai-catalyst","name":"RagaAI-Catalyst","tagline":"Python SDK for AI agent observability and evaluation","github_url":"https://github.com/raga-ai-hub/RagaAI-Catalyst","owner":"raga-ai-hub","repo":"RagaAI-Catalyst","owner_avatar_url":"https://avatars.githubusercontent.com/u/161833182?v=4","primary_language":"Python","stars":16148,"forks":3565,"topics":["agentic-ai","agentic-ai-development","agentneo","agents","ai-agent-monitoring","ai-application-debugging","ai-evaluation-tools","ai-performance-optimization","ai-tool-interaction-monitoring","llm-testing","llm-tracing","llmops"],"archived":false,"github_pushed_at":"2026-02-11T14:43:33+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/raga-ai-hub-ragaai-catalyst","markdown_url":"https://www.graphcanon.com/tools/raga-ai-hub-ragaai-catalyst.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/raga-ai-hub-ragaai-catalyst","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=raga-ai-hub-ragaai-catalyst","shared_categories":["evaluation-observability","ai-agents"]},{"slug":"shishirpatil-gorilla","name":"gorilla","tagline":"Training and Evaluating LLMs for Function Calls (Tool Calls)","github_url":"https://github.com/ShishirPatil/gorilla","owner":"ShishirPatil","repo":"gorilla","owner_avatar_url":"https://avatars.githubusercontent.com/u/30296397?v=4","primary_language":"Python","stars":12988,"forks":1397,"topics":["api","api-documentation","chatgpt","claude-api","gpt-4-api","llm","openai-api","openai-functions"],"archived":false,"github_pushed_at":"2026-04-13T03:19:45+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/shishirpatil-gorilla","markdown_url":"https://www.graphcanon.com/tools/shishirpatil-gorilla.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/shishirpatil-gorilla","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=shishirpatil-gorilla","shared_categories":["evaluation-observability"]},{"slug":"traceloop-openllmetry","name":"openllmetry","tagline":"Open-source observability for GenAI and LLM applications based on OpenTelemetry.","github_url":"https://github.com/traceloop/openllmetry","owner":"traceloop","repo":"openllmetry","owner_avatar_url":"https://avatars.githubusercontent.com/u/125419530?v=4","primary_language":"Python","stars":7377,"forks":1047,"topics":["artifical-intelligence","datascience","generative-ai","good-first-issue","good-first-issues","help-wanted","llm","llmops","metrics","ml","model-monitoring","monitoring","observability","open-source","open-telemetry","opentelemetry","opentelemetry-python","python"],"archived":false,"github_pushed_at":"2026-08-10T08:49:01+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/traceloop-openllmetry","markdown_url":"https://www.graphcanon.com/tools/traceloop-openllmetry.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/traceloop-openllmetry","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=traceloop-openllmetry","shared_categories":["evaluation-observability"]},{"slug":"stanford-crfm-helm","name":"helm","tagline":"Holistic, reproducible and transparent evaluation of foundation models","github_url":"https://github.com/stanford-crfm/helm","owner":"stanford-crfm","repo":"helm","owner_avatar_url":"https://avatars.githubusercontent.com/u/75054807?v=4","primary_language":"Python","stars":2873,"forks":406,"topics":[],"archived":false,"github_pushed_at":"2026-08-01T01:23:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/stanford-crfm-helm","markdown_url":"https://www.graphcanon.com/tools/stanford-crfm-helm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/stanford-crfm-helm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=stanford-crfm-helm","shared_categories":["evaluation-observability"]},{"slug":"openlit-openlit","name":"openlit","tagline":"A comprehensive open-source platform for AI Engineering with LLM Observability, Monitoring, and Management","github_url":"https://github.com/openlit/openlit","owner":"openlit","repo":"openlit","owner_avatar_url":"https://avatars.githubusercontent.com/u/149867240?v=4","primary_language":"TypeScript","stars":2664,"forks":342,"topics":["ai-observability","amd-gpu","clickhouse","distributed-tracing","genai","gpu-monitoring","grafana","langchain","llmops","llms","metrics","monitoring-tool","nvidia-smi","observability","open-source","openai","opentelemetry","otlp","python","tracing"],"archived":false,"github_pushed_at":"2026-07-31T18:39:37+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/openlit-openlit","markdown_url":"https://www.graphcanon.com/tools/openlit-openlit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openlit-openlit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openlit-openlit","shared_categories":["evaluation-observability"]},{"slug":"huggingface-lighteval","name":"lighteval","tagline":"All-in-one toolkit for evaluating LLMs across multiple backends","github_url":"https://github.com/huggingface/lighteval","owner":"huggingface","repo":"lighteval","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":2508,"forks":523,"topics":["evaluation","evaluation-framework","evaluation-metrics","huggingface"],"archived":false,"github_pushed_at":"2026-06-29T13:03:33+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huggingface-lighteval","markdown_url":"https://www.graphcanon.com/tools/huggingface-lighteval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-lighteval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-lighteval","shared_categories":["evaluation-observability"]},{"slug":"scale3-labs-langtrace","name":"langtrace","tagline":"Open Telemetry based observability tool for LLM applications","github_url":"https://github.com/Scale3-Labs/langtrace","owner":"Scale3-Labs","repo":"langtrace","owner_avatar_url":"https://avatars.githubusercontent.com/u/110545750?v=4","primary_language":"TypeScript","stars":1228,"forks":126,"topics":["ai","datasets","evaluations","gpt","langchain","llm","llm-framework","llmops","observability","open-source","open-telemetry","openai","prompt-engineering","tracing"],"archived":false,"github_pushed_at":"2025-11-17T15:08:48+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/scale3-labs-langtrace","markdown_url":"https://www.graphcanon.com/tools/scale3-labs-langtrace.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/scale3-labs-langtrace","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=scale3-labs-langtrace","shared_categories":["evaluation-observability"]},{"slug":"oxbshw-llm-agents-ecosystem-handbook","name":"LLM-Agents-Ecosystem-Handbook","tagline":"One-stop handbook for building, deploying, and understanding LLM agents","github_url":"https://github.com/oxbshw/LLM-Agents-Ecosystem-Handbook","owner":"oxbshw","repo":"LLM-Agents-Ecosystem-Handbook","owner_avatar_url":"https://avatars.githubusercontent.com/u/212214682?v=4","primary_language":"Python","stars":539,"forks":85,"topics":["ai","ai-agent","ai-agents","fine-tuning","finetuning-llms","freamework","llm","llmops","local-development","mcp-server","memory","rag","rag-chatbot","voice-agent"],"archived":false,"github_pushed_at":"2026-06-30T12:22:57+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/oxbshw-llm-agents-ecosystem-handbook","markdown_url":"https://www.graphcanon.com/tools/oxbshw-llm-agents-ecosystem-handbook.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/oxbshw-llm-agents-ecosystem-handbook","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=oxbshw-llm-agents-ecosystem-handbook","shared_categories":["evaluation-observability","ai-agents"]}]}}