{"data":{"node":{"slug":"langwatch-langevals","name":"langevals","tagline":"Provides a platform for evaluating and benchmarking LLM models using various evaluators","github_url":"https://github.com/langwatch/langevals","owner":"langwatch","repo":"langevals","owner_avatar_url":"https://avatars.githubusercontent.com/u/146763322?v=4","primary_language":null,"stars":72,"forks":10,"topics":["evaluation","guardrails","llm","openai"],"archived":true,"github_pushed_at":"2026-02-15T18:31:16+00:00","maintenance_label":"Archived","stars_delta_30d":0,"url":"https://www.graphcanon.com/tools/langwatch-langevals","markdown_url":"https://www.graphcanon.com/tools/langwatch-langevals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langwatch-langevals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langwatch-langevals"},"categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"evaluation","name":"evaluation"},{"slug":"guardrails","name":"guardrails"},{"slug":"llm","name":"llm"},{"slug":"openai","name":"openai"}],"edges":[],"neighbours":[{"slug":"confident-ai-deepeval","name":"deepeval","tagline":"LLM Evaluation Framework.","github_url":"https://github.com/confident-ai/deepeval","owner":"confident-ai","repo":"deepeval","owner_avatar_url":"https://avatars.githubusercontent.com/u/130858411?v=4","primary_language":"Python","stars":18342,"forks":1953,"topics":["evaluation-framework","evaluation-metrics","llm-evaluation","llm-evaluation-framework","llm-evaluation-metrics","python"],"archived":false,"github_pushed_at":"2026-09-18T17:06:58+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/confident-ai-deepeval","markdown_url":"https://www.graphcanon.com/tools/confident-ai-deepeval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/confident-ai-deepeval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=confident-ai-deepeval","shared_categories":["evaluation-observability"]},{"slug":"eleutherai-lm-evaluation-harness","name":"lm-evaluation-harness","tagline":"A framework for few-shot evaluation of language models.","github_url":"https://github.com/EleutherAI/lm-evaluation-harness","owner":"EleutherAI","repo":"lm-evaluation-harness","owner_avatar_url":"https://avatars.githubusercontent.com/u/68924597?v=4","primary_language":"Python","stars":13906,"forks":3547,"topics":["evaluation-framework","language-model","transformer"],"archived":false,"github_pushed_at":"2026-09-01T13:51:29+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness","markdown_url":"https://www.graphcanon.com/tools/eleutherai-lm-evaluation-harness.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eleutherai-lm-evaluation-harness","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eleutherai-lm-evaluation-harness","shared_categories":["evaluation-observability"]},{"slug":"andyyyy64-whichllm","name":"whichllm","tagline":"Command-line tool to find and benchmark local LLM performance","github_url":"https://github.com/Andyyyy64/whichllm","owner":"Andyyyy64","repo":"whichllm","owner_avatar_url":"https://avatars.githubusercontent.com/u/105579829?v=4","primary_language":"Python","stars":6666,"forks":368,"topics":["localllm"],"archived":false,"github_pushed_at":"2026-09-19T16:22:49+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/andyyyy64-whichllm","markdown_url":"https://www.graphcanon.com/tools/andyyyy64-whichllm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/andyyyy64-whichllm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=andyyyy64-whichllm","shared_categories":["evaluation-observability"]},{"slug":"meta-llama-purplellama","name":"PurpleLlama","tagline":"Set of tools to assess and improve LLM security","github_url":"https://github.com/meta-llama/PurpleLlama","owner":"meta-llama","repo":"PurpleLlama","owner_avatar_url":"https://avatars.githubusercontent.com/u/153379578?v=4","primary_language":"Python","stars":4380,"forks":769,"topics":[],"archived":false,"github_pushed_at":"2026-08-18T01:15:09+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/meta-llama-purplellama","markdown_url":"https://www.graphcanon.com/tools/meta-llama-purplellama.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/meta-llama-purplellama","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=meta-llama-purplellama","shared_categories":["evaluation-observability"]},{"slug":"evolvinglmms-lab-lmms-eval","name":"lmms-eval","tagline":"One-for-All Multimodal Evaluation Toolkit Across Text, Image, Video, and Audio Tasks","github_url":"https://github.com/EvolvingLMMs-Lab/lmms-eval","owner":"EvolvingLMMs-Lab","repo":"lmms-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/154951679?v=4","primary_language":"Python","stars":4368,"forks":639,"topics":["agi","audio-evaluation","benchmark","evaluation","large-language-models","llm-evaluation","multimodal","multimodal-evaluation","video-understanding","vision-language-model","vlm"],"archived":false,"github_pushed_at":"2026-08-06T02:22:23+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/evolvinglmms-lab-lmms-eval","markdown_url":"https://www.graphcanon.com/tools/evolvinglmms-lab-lmms-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evolvinglmms-lab-lmms-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evolvinglmms-lab-lmms-eval","shared_categories":["evaluation-observability"]},{"slug":"theopenco-llmgateway","name":"llmgateway","tagline":"Route and manage LLM requests via unified API interface","github_url":"https://github.com/theopenco/llmgateway","owner":"theopenco","repo":"llmgateway","owner_avatar_url":"https://avatars.githubusercontent.com/u/211671860?v=4","primary_language":"TypeScript","stars":1624,"forks":181,"topics":["ai","ai-gateway","analytics","anthropic","api-key-management","claude","codex","enterprise","guardrails","inference","llm","llm-gateway","llm-proxy","llms","observability","openai","opencode","rate-limiting","typescript"],"archived":false,"github_pushed_at":"2026-09-11T06:00:04+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/theopenco-llmgateway","markdown_url":"https://www.graphcanon.com/tools/theopenco-llmgateway.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/theopenco-llmgateway","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=theopenco-llmgateway","shared_categories":[]},{"slug":"georgian-io-llm-finetuning-toolkit","name":"LLM-Finetuning-Toolkit","tagline":"Toolkit for fine-tuning and testing open-source large language models","github_url":"https://github.com/georgian-io/LLM-Finetuning-Toolkit","owner":"georgian-io","repo":"LLM-Finetuning-Toolkit","owner_avatar_url":"https://avatars.githubusercontent.com/u/10764713?v=4","primary_language":"Python","stars":870,"forks":107,"topics":["ablation-study","classification","falcon","fine-tuning","finetuning","flan-t5","large-language-models","llama2","llm-test","lora","mistral-7b","nlp","nlp-machine-learning","qlora","redpajama","summarization","unit-testing","zephyr"],"archived":false,"github_pushed_at":"2026-05-04T16:33:40+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/georgian-io-llm-finetuning-toolkit","markdown_url":"https://www.graphcanon.com/tools/georgian-io-llm-finetuning-toolkit.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/georgian-io-llm-finetuning-toolkit","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=georgian-io-llm-finetuning-toolkit","shared_categories":[]},{"slug":"declare-lab-instruct-eval","name":"instruct-eval","tagline":"Quantitative evaluation for instruction-tuned language models","github_url":"https://github.com/declare-lab/instruct-eval","owner":"declare-lab","repo":"instruct-eval","owner_avatar_url":"https://avatars.githubusercontent.com/u/59164695?v=4","primary_language":"Python","stars":553,"forks":45,"topics":["instruct-tuning","llm"],"archived":false,"github_pushed_at":"2024-03-10T05:00:00+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval","markdown_url":"https://www.graphcanon.com/tools/declare-lab-instruct-eval.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/declare-lab-instruct-eval","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=declare-lab-instruct-eval","shared_categories":["evaluation-observability"]},{"slug":"curated-awesome-lists-awesome-llms-fine-tuning","name":"awesome-llms-fine-tuning","tagline":"A comprehensive collection of resources for fine-tuning Large Language Models.","github_url":"https://github.com/Curated-Awesome-Lists/awesome-llms-fine-tuning","owner":"Curated-Awesome-Lists","repo":"awesome-llms-fine-tuning","owner_avatar_url":"https://avatars.githubusercontent.com/u/142611331?v=4","primary_language":null,"stars":527,"forks":80,"topics":["ai","awesome-list","deep-learning","fine-tuning","gpt","large-language-models","llms","machine-learning","nlp","transformers"],"archived":false,"github_pushed_at":"2026-09-04T19:22:27+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/curated-awesome-lists-awesome-llms-fine-tuning","markdown_url":"https://www.graphcanon.com/tools/curated-awesome-lists-awesome-llms-fine-tuning.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/curated-awesome-lists-awesome-llms-fine-tuning","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=curated-awesome-lists-awesome-llms-fine-tuning","shared_categories":[]},{"slug":"jonathanchaveztamales-llm-leaderboard","name":"llm-leaderboard","tagline":"Comprehensive LLM benchmark scores and provider prices","github_url":"https://github.com/JonathanChavezTamales/llm-leaderboard","owner":"JonathanChavezTamales","repo":"llm-leaderboard","owner_avatar_url":"https://avatars.githubusercontent.com/u/22694942?v=4","primary_language":"JavaScript","stars":356,"forks":40,"topics":["llm","llm-agents","llm-evaluation","llmops","llms-benchmarking"],"archived":false,"github_pushed_at":"2025-10-24T17:47:59+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/jonathanchaveztamales-llm-leaderboard","markdown_url":"https://www.graphcanon.com/tools/jonathanchaveztamales-llm-leaderboard.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jonathanchaveztamales-llm-leaderboard","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jonathanchaveztamales-llm-leaderboard","shared_categories":["evaluation-observability"]},{"slug":"verifywise-ai-verifywise","name":"verifywise","tagline":"Complete AI governance and LLM Evals platform","github_url":"https://github.com/verifywise-ai/verifywise","owner":"verifywise-ai","repo":"verifywise","owner_avatar_url":"https://avatars.githubusercontent.com/u/262239808?v=4","primary_language":"TypeScript","stars":354,"forks":117,"topics":["ai","ai-auditing","ai-compliance","ai-governance","ai-governance-model","ai-risk","audit","auditing","compliance","eu-ai-act","governance","grc","iso27001","iso42001","llm-eval","llm-evaluation","nist-ai-rmf","risk-management"],"archived":false,"github_pushed_at":"2026-09-19T19:03:23+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/verifywise-ai-verifywise","markdown_url":"https://www.graphcanon.com/tools/verifywise-ai-verifywise.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/verifywise-ai-verifywise","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=verifywise-ai-verifywise","shared_categories":["evaluation-observability"]},{"slug":"athina-ai-athina-evals","name":"athina-evals","tagline":"Python SDK for evaluating LLM generated responses","github_url":"https://github.com/athina-ai/athina-evals","owner":"athina-ai","repo":"athina-evals","owner_avatar_url":"https://avatars.githubusercontent.com/u/139258696?v=4","primary_language":"Python","stars":301,"forks":22,"topics":["evaluation","evaluation-framework","evaluation-metrics","llm-eval","llm-evaluation","llm-evaluation-toolkit","llm-ops","llmops"],"archived":false,"github_pushed_at":"2025-06-06T15:54:38+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/athina-ai-athina-evals","markdown_url":"https://www.graphcanon.com/tools/athina-ai-athina-evals.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/athina-ai-athina-evals","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=athina-ai-athina-evals","shared_categories":["evaluation-observability"]}]}}