{"data":{"slug":"predibase-lorax","name":"lorax","tagline":"Multi-LoRA inference server for scalable fine-tuned LLMs","github_url":"https://github.com/predibase/lorax","owner":"predibase","repo":"lorax","owner_avatar_url":"https://avatars.githubusercontent.com/u/75280641?v=4","primary_language":"Python","stars":3826,"forks":326,"topics":["fine-tuning","gpt","llama","llm","llm-inference","llm-serving","llmops","lora","model-serving","pytorch","transformers"],"archived":false,"github_pushed_at":"2026-05-28T18:12:20+00:00","maintenance_label":"Steady","stars_delta_30d":10,"url":"https://www.graphcanon.com/tools/predibase-lorax","markdown_url":"https://www.graphcanon.com/tools/predibase-lorax.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/predibase-lorax","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=predibase-lorax","description":"Multi-LoRA inference server that scales to 1000s of fine-tuned LLMs","homepage_url":"https://loraexchange.ai","license":"Apache-2.0","open_issues":185,"watchers":33,"ai_summary":"Lorax is a Python-based multi-LoRA inference server designed to handle thousands of fine-tuned language models, utilizing PyTorch and transformers. It requires an Nvidia GPU with compatible CUDA drivers.","readme_excerpt":"## 🏃‍♂️ Getting Started\n\nWe recommend starting with our pre-built Docker image to avoid compiling custom CUDA kernels and other dependencies.\n\n---\n\n### Requirements\n\nThe minimum system requirements need to run LoRAX include:\n\n- Nvidia GPU (Ampere generation or above)\n- CUDA 11.8 compatible device drivers and above\n- Linux OS\n- Docker (for this guide)","github_created_at":"2023-10-20T18:19:49+00:00","created_at":"2026-07-07T17:42:15.234047+00:00","updated_at":"2026-08-20T12:01:54.308185+00:00","categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"}],"tags":[{"slug":"fine-tuning","name":"fine-tuning"},{"slug":"gpt","name":"gpt"},{"slug":"llama","name":"llama"},{"slug":"llm-inference","name":"llm-inference"},{"slug":"llm-serving","name":"llm-serving"},{"slug":"pytorch","name":"pytorch"},{"slug":"transformers","name":"transformers"}],"trust":{"provenance":{"is_fork":false,"github_id":707818217,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-08-20T12:01:53.413Z","maintenance":{"label":"Steady","score":60,"methodology":"github_public_v1","releases_90d":0,"days_since_push":83,"last_release_at":"2025-01-13T23:12:10Z","stars_delta_30d":10,"open_issues_delta_30d":1},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-11T11:20:27.348Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-08-20T12:01:53.958Z"},"deploy":{"source":"dockerfile:Dockerfile","self_host":true,"observed_at":"2026-08-20T12:01:53.958Z","managed_saas":false},"languages":{"value":["python"],"source":"github.language","observed_at":"2026-08-20T12:01:53.958Z"},"has_docker":{"value":true,"source":"dockerfile:Dockerfile","observed_at":"2026-08-20T12:01:53.958Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-08-20T12:01:53.958Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":{"notes":["Requires Nvidia GPU (Ampere generation or above)","CUDA 11.8 compatible drivers and higher","Linux OS required","Docker for setup"],"min_ram_gb":null},"constraints":{"min_ram_gb":null},"when_to_use":["- You require an infrastructure that can manage up to thousands of LoRA-adapted LLMs simultaneously for high-throughput inference.","- Your environment includes Nvidia GPUs from the Ampere generation or above, as it requires significant computing power. Specifically, Lorax leverages CUDA 11.8 for these heavy compute tasks.","- You are working with Python and need a solution that natively supports PyTorch and transformers."],"when_not_to_use":["- Your system does not meet the minimum hardware requirements (Nvidia Ampere generation GPU or higher).","- If your team lacks experience with Docker and Linux-based systems since Lorax's setup guidelines rely heavily on these technologies.","- You are restricted to software licenses other than Apache-2.0, as Lorax is distributed under this specific license."],"source":"enrich:decision_facts","observed_at":"2026-07-11T02:44:10.202Z"},"constraint_facets":{"min_ram_gb":null},"decision_summary":[{"label":"Requirements","value":"Requires Nvidia GPU (Ampere generation or above); CUDA 11.8 compatible drivers and higher; Linux OS required; Docker for setup"},{"label":"Adopt for","value":"Lorax is a Python-based inference server specialized in managing large fleets of LoRA-adapted language models, which can scale up to thousands of fine-tuned LLMs. It supports platforms like GPT and LLaMA using PyTorch."}]}}