{"data":{"node":{"slug":"joyce94-llm-rlhf-tuning","name":"LLM-RLHF-Tuning","tagline":"LLM Tuning with PEFT (SFT+RM+PPO+DPO with LoRA)","github_url":"https://github.com/Joyce94/LLM-RLHF-Tuning","owner":"Joyce94","repo":"LLM-RLHF-Tuning","owner_avatar_url":"https://avatars.githubusercontent.com/u/28557140?v=4","primary_language":"Python","stars":452,"forks":24,"topics":["fine-tuning","language-model","llama","llm","lora","peft","ppo","reinforcement-learning","rlhf"],"archived":false,"github_pushed_at":"2023-10-11T08:41:20+00:00","maintenance_label":"Dormant","stars_delta_30d":-1,"url":"https://www.graphcanon.com/tools/joyce94-llm-rlhf-tuning","markdown_url":"https://www.graphcanon.com/tools/joyce94-llm-rlhf-tuning.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/joyce94-llm-rlhf-tuning","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=joyce94-llm-rlhf-tuning"},"categories":[{"slug":"llm-frameworks","name":"LLM Frameworks","url":"https://www.graphcanon.com/categories/llm-frameworks","markdown_url":"https://www.graphcanon.com/categories/llm-frameworks.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/llm-frameworks"},{"slug":"model-training","name":"Model Training","url":"https://www.graphcanon.com/categories/model-training","markdown_url":"https://www.graphcanon.com/categories/model-training.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/model-training"}],"tags":[{"slug":"fine-tuning","name":"fine-tuning"},{"slug":"language-model","name":"language-model"},{"slug":"llama","name":"llama"},{"slug":"llm","name":"llm"},{"slug":"lora","name":"lora"},{"slug":"peft","name":"peft"},{"slug":"ppo","name":"ppo"},{"slug":"reinforcement-learning","name":"reinforcement-learning"}],"edges":[],"neighbours":[{"slug":"huggingface-peft","name":"peft","tagline":"State-of-the-art Parameter-Efficient Fine-Tuning","github_url":"https://github.com/huggingface/peft","owner":"huggingface","repo":"peft","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":21585,"forks":2446,"topics":["adapter","diffusion","fine-tuning","llm","lora","parameter-efficient-learning","peft","python","pytorch","transformers"],"archived":false,"github_pushed_at":"2026-08-22T01:15:06+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/huggingface-peft","markdown_url":"https://www.graphcanon.com/tools/huggingface-peft.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-peft","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-peft","shared_categories":["model-training","llm-frameworks"]},{"slug":"huggingface-trl","name":"trl","tagline":"Train transformer language models with reinforcement learning.","github_url":"https://github.com/huggingface/trl","owner":"huggingface","repo":"trl","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":19016,"forks":2891,"topics":[],"archived":false,"github_pushed_at":"2026-08-06T10:02:43+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-trl","markdown_url":"https://www.graphcanon.com/tools/huggingface-trl.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-trl","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-trl","shared_categories":["model-training"]},{"slug":"tloen-alpaca-lora","name":"alpaca-lora","tagline":"Instruct-tune LLaMA on consumer hardware","github_url":"https://github.com/tloen/alpaca-lora","owner":"tloen","repo":"alpaca-lora","owner_avatar_url":"https://avatars.githubusercontent.com/u/4811103?v=4","primary_language":"Jupyter Notebook","stars":18912,"forks":2180,"topics":[],"archived":false,"github_pushed_at":"2024-07-29T13:37:49+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/tloen-alpaca-lora","markdown_url":"https://www.graphcanon.com/tools/tloen-alpaca-lora.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tloen-alpaca-lora","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tloen-alpaca-lora","shared_categories":["model-training","llm-frameworks"]},{"slug":"lightning-ai-litgpt","name":"litgpt","tagline":"High-performance LLMs with recipes for pretraining, finetuning and deployment","github_url":"https://github.com/Lightning-AI/litgpt","owner":"Lightning-AI","repo":"litgpt","owner_avatar_url":"https://avatars.githubusercontent.com/u/58386951?v=4","primary_language":"Python","stars":13605,"forks":1483,"topics":["ai","artificial-intelligence","deep-learning","large-language-models","llm","llm-inference","llms"],"archived":false,"github_pushed_at":"2026-07-20T10:24:12+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/lightning-ai-litgpt","markdown_url":"https://www.graphcanon.com/tools/lightning-ai-litgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/lightning-ai-litgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=lightning-ai-litgpt","shared_categories":["model-training","llm-frameworks"]},{"slug":"artidoro-qlora","name":"qlora","tagline":"QLoRA finetuning of quantized LLMs","github_url":"https://github.com/artidoro/qlora","owner":"artidoro","repo":"qlora","owner_avatar_url":"https://avatars.githubusercontent.com/u/11949572?v=4","primary_language":"Jupyter Notebook","stars":10979,"forks":876,"topics":[],"archived":false,"github_pushed_at":"2024-06-10T19:20:16+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/artidoro-qlora","markdown_url":"https://www.graphcanon.com/tools/artidoro-qlora.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/artidoro-qlora","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=artidoro-qlora","shared_categories":["model-training","llm-frameworks"]},{"slug":"opendilab-awesome-rlhf","name":"awesome-RLHF","tagline":"A curated list of reinforcement learning with human feedback resources (continually updated)","github_url":"https://github.com/opendilab/awesome-RLHF","owner":"opendilab","repo":"awesome-RLHF","owner_avatar_url":"https://avatars.githubusercontent.com/u/86840398?v=4","primary_language":null,"stars":4422,"forks":258,"topics":["deep-learning","deep-reinforcement-learning","human-feedback","large-language-models","reinforcement-learning","rlhf"],"archived":false,"github_pushed_at":"2026-05-20T12:56:15+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/opendilab-awesome-rlhf","markdown_url":"https://www.graphcanon.com/tools/opendilab-awesome-rlhf.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/opendilab-awesome-rlhf","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=opendilab-awesome-rlhf","shared_categories":["model-training"]},{"slug":"alibaba-roll","name":"ROLL","tagline":"Scaling Library for Reinforcement Learning with Large Language Models","github_url":"https://github.com/alibaba/ROLL","owner":"alibaba","repo":"ROLL","owner_avatar_url":"https://avatars.githubusercontent.com/u/1961952?v=4","primary_language":"Python","stars":3354,"forks":304,"topics":["agentic","rlhf","rlvr"],"archived":false,"github_pushed_at":"2026-08-07T01:51:58+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/alibaba-roll","markdown_url":"https://www.graphcanon.com/tools/alibaba-roll.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alibaba-roll","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alibaba-roll","shared_categories":["model-training"]},{"slug":"eladlev-autoprompt","name":"AutoPrompt","tagline":"Framework for prompt tuning using Intent-based Prompt Calibration","github_url":"https://github.com/Eladlev/AutoPrompt","owner":"Eladlev","repo":"AutoPrompt","owner_avatar_url":"https://avatars.githubusercontent.com/u/28984104?v=4","primary_language":"Python","stars":2993,"forks":264,"topics":["prompt-engineering","prompt-tuning","synthetic-dataset-generation"],"archived":false,"github_pushed_at":"2025-12-02T17:23:20+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/eladlev-autoprompt","markdown_url":"https://www.graphcanon.com/tools/eladlev-autoprompt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/eladlev-autoprompt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=eladlev-autoprompt","shared_categories":["llm-frameworks"]},{"slug":"ashishpatel26-llm-finetuning","name":"LLM-Finetuning","tagline":"LLM Finetuning with PEFT","github_url":"https://github.com/ashishpatel26/LLM-Finetuning","owner":"ashishpatel26","repo":"LLM-Finetuning","owner_avatar_url":"https://avatars.githubusercontent.com/u/3095771?v=4","primary_language":"Jupyter Notebook","stars":2979,"forks":771,"topics":["falcon","fine-tuning","huggingface","llama","llama2","llm","llms","lora","peft","pytorch","text-generation"],"archived":false,"github_pushed_at":"2025-08-01T12:00:20+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/ashishpatel26-llm-finetuning","markdown_url":"https://www.graphcanon.com/tools/ashishpatel26-llm-finetuning.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ashishpatel26-llm-finetuning","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ashishpatel26-llm-finetuning","shared_categories":["model-training","llm-frameworks"]},{"slug":"thudm-p-tuning-v2","name":"P-tuning-v2","tagline":"Optimized deep prompt tuning strategy comparable to fine-tuning across scales and tasks","github_url":"https://github.com/THUDM/P-tuning-v2","owner":"THUDM","repo":"P-tuning-v2","owner_avatar_url":"https://avatars.githubusercontent.com/u/48590610?v=4","primary_language":"Python","stars":2077,"forks":213,"topics":["natural-language-processing","p-tuning","parameter-efficient-learning","pretrained-language-model","prompt-tuning"],"archived":false,"github_pushed_at":"2023-11-16T04:38:09+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/thudm-p-tuning-v2","markdown_url":"https://www.graphcanon.com/tools/thudm-p-tuning-v2.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/thudm-p-tuning-v2","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=thudm-p-tuning-v2","shared_categories":["model-training"]},{"slug":"sakanaai-text-to-lora","name":"text-to-lora","tagline":"Hypernetworks for adapting LLMs to specific tasks via textual descriptions","github_url":"https://github.com/SakanaAI/text-to-lora","owner":"SakanaAI","repo":"text-to-lora","owner_avatar_url":"https://avatars.githubusercontent.com/u/140988036?v=4","primary_language":"Python","stars":1300,"forks":88,"topics":["fine-tuning","hypernetworks","llm","lora","machine-learning"],"archived":false,"github_pushed_at":"2025-06-08T14:42:10+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/sakanaai-text-to-lora","markdown_url":"https://www.graphcanon.com/tools/sakanaai-text-to-lora.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/sakanaai-text-to-lora","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=sakanaai-text-to-lora","shared_categories":["model-training"]},{"slug":"agi-edgerunners-llm-adapters","name":"LLM-Adapters","tagline":"Code for EMNLP 2023 Paper on Parameter-Efficient Fine-Tuning of LLMs","github_url":"https://github.com/AGI-Edgerunners/LLM-Adapters","owner":"AGI-Edgerunners","repo":"LLM-Adapters","owner_avatar_url":"https://avatars.githubusercontent.com/u/128879899?v=4","primary_language":"Python","stars":1233,"forks":115,"topics":["adapters","fine-tuning","large-language-models","parameter-efficient"],"archived":false,"github_pushed_at":"2024-03-10T08:20:15+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/agi-edgerunners-llm-adapters","markdown_url":"https://www.graphcanon.com/tools/agi-edgerunners-llm-adapters.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/agi-edgerunners-llm-adapters","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=agi-edgerunners-llm-adapters","shared_categories":["model-training","llm-frameworks"]}]}}