{"data":{"node":{"slug":"uclaml-sppo","name":"SPPO","tagline":"Official implementation of Self-Play Preference Optimization for fine-tuning large language models via RLHF","github_url":"https://github.com/uclaml/SPPO","owner":"uclaml","repo":"SPPO","owner_avatar_url":"https://avatars.githubusercontent.com/u/22385378?v=4","primary_language":"Python","stars":589,"forks":48,"topics":["deep-learning","fine-tuning","large-language-models","rlhf","self-play"],"archived":false,"github_pushed_at":"2025-01-23T01:25:48+00:00","maintenance_label":"Dormant","stars_delta_30d":-1,"url":"https://www.graphcanon.com/tools/uclaml-sppo","markdown_url":"https://www.graphcanon.com/tools/uclaml-sppo.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/uclaml-sppo","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=uclaml-sppo"},"categories":[{"slug":"llm-frameworks","name":"LLM Frameworks","url":"https://www.graphcanon.com/categories/llm-frameworks","markdown_url":"https://www.graphcanon.com/categories/llm-frameworks.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/llm-frameworks"},{"slug":"model-training","name":"Model Training","url":"https://www.graphcanon.com/categories/model-training","markdown_url":"https://www.graphcanon.com/categories/model-training.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/model-training"}],"tags":[{"slug":"deep-learning","name":"deep-learning"},{"slug":"fine-tuning","name":"fine-tuning"},{"slug":"large-language-models","name":"large language models"},{"slug":"rlhf","name":"rlhf"},{"slug":"self-play","name":"self-play"}],"edges":[],"neighbours":[{"slug":"optuna-optuna","name":"optuna","tagline":"A hyperparameter optimization framework","github_url":"https://github.com/optuna/optuna","owner":"optuna","repo":"optuna","owner_avatar_url":"https://avatars.githubusercontent.com/u/57251745?v=4","primary_language":"Python","stars":14603,"forks":1361,"topics":["distributed","hyperparameter-optimization","machine-learning","parallel","python"],"archived":false,"github_pushed_at":"2026-08-03T06:38:06+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/optuna-optuna","markdown_url":"https://www.graphcanon.com/tools/optuna-optuna.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/optuna-optuna","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=optuna-optuna","shared_categories":["model-training"]},{"slug":"pyspur-dev-pyspur","name":"pyspur","tagline":"A visual playground for agentic workflows","github_url":"https://github.com/PySpur-Dev/pyspur","owner":"PySpur-Dev","repo":"pyspur","owner_avatar_url":"https://avatars.githubusercontent.com/u/182547524?v=4","primary_language":"TypeScript","stars":5771,"forks":429,"topics":["agent","agents","ai","builder","deepseek","framework","gemini","graph","human-in-the-loop","llm","llms","loops","multimodal","ollama","python","rag","reasoning","tool","trace","workflow"],"archived":false,"github_pushed_at":"2026-06-29T17:53:12+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/pyspur-dev-pyspur","markdown_url":"https://www.graphcanon.com/tools/pyspur-dev-pyspur.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pyspur-dev-pyspur","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pyspur-dev-pyspur","shared_categories":[]},{"slug":"opendilab-awesome-rlhf","name":"awesome-RLHF","tagline":"A curated list of reinforcement learning with human feedback resources (continually updated)","github_url":"https://github.com/opendilab/awesome-RLHF","owner":"opendilab","repo":"awesome-RLHF","owner_avatar_url":"https://avatars.githubusercontent.com/u/86840398?v=4","primary_language":null,"stars":4422,"forks":258,"topics":["deep-learning","deep-reinforcement-learning","human-feedback","large-language-models","reinforcement-learning","rlhf"],"archived":false,"github_pushed_at":"2026-05-20T12:56:15+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/opendilab-awesome-rlhf","markdown_url":"https://www.graphcanon.com/tools/opendilab-awesome-rlhf.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/opendilab-awesome-rlhf","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=opendilab-awesome-rlhf","shared_categories":["model-training"]},{"slug":"alibaba-roll","name":"ROLL","tagline":"Scaling Library for Reinforcement Learning with Large Language Models","github_url":"https://github.com/alibaba/ROLL","owner":"alibaba","repo":"ROLL","owner_avatar_url":"https://avatars.githubusercontent.com/u/1961952?v=4","primary_language":"Python","stars":3354,"forks":304,"topics":["agentic","rlhf","rlvr"],"archived":false,"github_pushed_at":"2026-08-07T01:51:58+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/alibaba-roll","markdown_url":"https://www.graphcanon.com/tools/alibaba-roll.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alibaba-roll","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alibaba-roll","shared_categories":["model-training"]},{"slug":"uclaml-spin","name":"SPIN","tagline":"Official implementation of Self-Play Fine-Tuning","github_url":"https://github.com/uclaml/SPIN","owner":"uclaml","repo":"SPIN","owner_avatar_url":"https://avatars.githubusercontent.com/u/22385378?v=4","primary_language":"Python","stars":1254,"forks":106,"topics":["deep-learning","fine-tuning","large-language-models","self-play"],"archived":false,"github_pushed_at":"2024-05-08T05:59:37+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/uclaml-spin","markdown_url":"https://www.graphcanon.com/tools/uclaml-spin.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/uclaml-spin","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=uclaml-spin","shared_categories":["model-training","llm-frameworks"]},{"slug":"salesforce-coderl","name":"CodeRL","tagline":"CodeRL: Combines pretrained models and reinforcement learning for code generation.","github_url":"https://github.com/salesforce/CodeRL","owner":"salesforce","repo":"CodeRL","owner_avatar_url":"https://avatars.githubusercontent.com/u/453694?v=4","primary_language":"Python","stars":574,"forks":69,"topics":["ai","codegeneration","languagemodel","machinelearning","programsynthesis","reinforcementlearning"],"archived":false,"github_pushed_at":"2026-06-02T18:14:33+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/salesforce-coderl","markdown_url":"https://www.graphcanon.com/tools/salesforce-coderl.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/salesforce-coderl","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=salesforce-coderl","shared_categories":["model-training"]},{"slug":"sponsiolabs-sponsio","name":"Sponsio","tagline":"Deterministic safety solutions for probabilistic AI agents","github_url":"https://github.com/SponsioLabs/Sponsio","owner":"SponsioLabs","repo":"Sponsio","owner_avatar_url":"https://avatars.githubusercontent.com/u/270103425?v=4","primary_language":"Python","stars":469,"forks":28,"topics":["agent-guardrails","agent-harness","agent-runtime","agent-safety","agent-security","agent-skills","agentic-ai","ai-agents","ai-security","deterministic","guardrails","intent-verification","mcp","open-source","openclaw","openclaw-skills","policy-engine","prompt-injection","runtime-safety","self-hosted"],"archived":false,"github_pushed_at":"2026-08-07T21:25:35+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/sponsiolabs-sponsio","markdown_url":"https://www.graphcanon.com/tools/sponsiolabs-sponsio.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/sponsiolabs-sponsio","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=sponsiolabs-sponsio","shared_categories":[]},{"slug":"joyce94-llm-rlhf-tuning","name":"LLM-RLHF-Tuning","tagline":"LLM Tuning with PEFT (SFT+RM+PPO+DPO with LoRA)","github_url":"https://github.com/Joyce94/LLM-RLHF-Tuning","owner":"Joyce94","repo":"LLM-RLHF-Tuning","owner_avatar_url":"https://avatars.githubusercontent.com/u/28557140?v=4","primary_language":"Python","stars":452,"forks":24,"topics":["fine-tuning","language-model","llama","llm","lora","peft","ppo","reinforcement-learning","rlhf"],"archived":false,"github_pushed_at":"2023-10-11T08:41:20+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/joyce94-llm-rlhf-tuning","markdown_url":"https://www.graphcanon.com/tools/joyce94-llm-rlhf-tuning.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/joyce94-llm-rlhf-tuning","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=joyce94-llm-rlhf-tuning","shared_categories":["model-training","llm-frameworks"]},{"slug":"reddy-lab-code-research-ppocoder","name":"PPOCoder","tagline":"PPOCoder utilizes deep reinforcement learning for execution-based code generation","github_url":"https://github.com/reddy-lab-code-research/PPOCoder","owner":"reddy-lab-code-research","repo":"PPOCoder","owner_avatar_url":"https://avatars.githubusercontent.com/u/99287607?v=4","primary_language":"Python","stars":116,"forks":12,"topics":["code-generation","deep-reinforcement-learning","language-model","programming-language"],"archived":false,"github_pushed_at":"2024-01-09T21:03:07+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/reddy-lab-code-research-ppocoder","markdown_url":"https://www.graphcanon.com/tools/reddy-lab-code-research-ppocoder.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/reddy-lab-code-research-ppocoder","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=reddy-lab-code-research-ppocoder","shared_categories":["model-training"]},{"slug":"kolenaio-autoarena","name":"autoarena","tagline":"Automated evaluation of LLMs and RAG systems","github_url":"https://github.com/kolenaIO/autoarena","owner":"kolenaIO","repo":"autoarena","owner_avatar_url":"https://avatars.githubusercontent.com/u/77010818?v=4","primary_language":"TypeScript","stars":108,"forks":9,"topics":["ai","evaluation","hacktoberfest","llm","llm-evaluation","rag","testing"],"archived":false,"github_pushed_at":"2024-12-16T12:25:44+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/kolenaio-autoarena","markdown_url":"https://www.graphcanon.com/tools/kolenaio-autoarena.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kolenaio-autoarena","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kolenaio-autoarena","shared_categories":[]},{"slug":"future-agi-agent-opt","name":"agent-opt","tagline":"Open Source Library for Automated Optimization of AI Agent Workflows","github_url":"https://github.com/future-agi/agent-opt","owner":"future-agi","repo":"agent-opt","owner_avatar_url":"https://avatars.githubusercontent.com/u/147392366?v=4","primary_language":"Python","stars":71,"forks":7,"topics":["agent","ai-agents","aioptimization","automation","cicd","evaluation","optimization-algorithms"],"archived":false,"github_pushed_at":"2026-06-30T00:10:45+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/future-agi-agent-opt","markdown_url":"https://www.graphcanon.com/tools/future-agi-agent-opt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/future-agi-agent-opt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=future-agi-agent-opt","shared_categories":[]}]}}