{"data":{"slug":"argilla-io-distilabel","name":"distilabel","tagline":"Framework for synthetic data and AI feedback pipelines","github_url":"https://github.com/argilla-io/distilabel","owner":"argilla-io","repo":"distilabel","owner_avatar_url":"https://avatars.githubusercontent.com/u/18415507?v=4","primary_language":"Python","stars":3353,"forks":252,"topics":["ai","huggingface","llms","openai","python","rlaif","rlhf","synthetic-data","synthetic-dataset-generation"],"archived":false,"github_pushed_at":"2026-07-27T21:55:31+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/argilla-io-distilabel","markdown_url":"https://www.graphcanon.com/tools/argilla-io-distilabel.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/argilla-io-distilabel","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=argilla-io-distilabel","description":"Distilabel is a framework for synthetic data and AI feedback for engineers who need fast, reliable and scalable pipelines based on verified research papers.","homepage_url":"https://distilabel.argilla.io","license":"Apache-2.0","open_issues":102,"watchers":23,"ai_summary":"Distilabel provides engineers with tools to create fast, reliable, and scalable pipelines using verified research papers, focusing on the generation of synthetic datasets.","readme_excerpt":"## Installation\n\n```sh\npip install distilabel --upgrade\n```\n\nRequires Python 3.9+\n\nIn addition, the following extras are available:","github_created_at":"2023-10-16T14:12:33+00:00","created_at":"2026-07-11T23:28:39.705324+00:00","updated_at":"2026-08-03T18:00:40.839389+00:00","categories":[{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"},{"slug":"model-training","name":"Model Training","url":"https://www.graphcanon.com/categories/model-training","markdown_url":"https://www.graphcanon.com/categories/model-training.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/model-training"}],"tags":[{"slug":"ai","name":"ai"},{"slug":"huggingface","name":"huggingface"},{"slug":"llms","name":"llms"},{"slug":"openai","name":"openai"},{"slug":"python","name":"python"},{"slug":"rlaif","name":"rlaif"},{"slug":"rlhf","name":"rlhf"},{"slug":"synthetic-data","name":"synthetic-data"}],"trust":{"provenance":{"is_fork":false,"github_id":705695450,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-08-03T18:00:39.981Z","maintenance":{"label":"Very active","score":96,"methodology":"github_public_v1","releases_90d":0,"days_since_push":6,"last_release_at":"2025-01-28T10:08:05Z"},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-11T23:28:41.500Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-08-03T18:00:40.494Z"},"has_cli":{"value":true,"source":"pyproject.toml:[project.scripts]","observed_at":"2026-08-03T18:00:40.494Z"},"languages":{"value":["python"],"source":"github.language+pyproject.toml","observed_at":"2026-08-03T18:00:40.494Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-08-03T18:00:40.494Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":null,"constraints":null,"when_to_use":["When you need to work with scalable and high-reliability pipelines backed by rigorous academic research.","If your project requires integration capabilities with frameworks such as Hugging Face, OpenAI, or others in the domain of large language models (LLMs)."],"when_not_to_use":["For projects that prioritize immediate availability over the rigor of using research-verified methods for synthetic data creation.","If your technical environment does not comply with Python 3.9+ requirement and additional dependencies required to run Distilabel."],"source":"enrich:decision_facts","observed_at":"2026-07-17T12:55:18.990Z"},"constraint_facets":null,"decision_summary":[{"label":"Adopt for","value":"Distilabel is designed to offer engineers tools focusing on synthetic dataset generation and fast feedback pipelines based on validated research."}]}}