{"data":{"slug":"ucbepic-docetl","name":"docetl","tagline":"A system for agentic LLM-powered data processing and ETL","github_url":"https://github.com/ucbepic/docetl","owner":"ucbepic","repo":"docetl","owner_avatar_url":"https://avatars.githubusercontent.com/u/88680502?v=4","primary_language":"Python","stars":4092,"forks":443,"topics":["agents","data","data-pipelines","document-analysis","document-processing","elt","etl","llm","python","semantic-data","unstructured-data","unstructured-data-analysis","workflow"],"archived":false,"github_pushed_at":"2026-09-05T16:55:18+00:00","maintenance_label":"Active","stars_delta_30d":131,"url":"https://www.graphcanon.com/tools/ucbepic-docetl","markdown_url":"https://www.graphcanon.com/tools/ucbepic-docetl.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ucbepic-docetl","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ucbepic-docetl","description":"A system for agentic LLM-powered data processing and ETL","homepage_url":"https://docetl.org","license":"MIT","open_issues":45,"watchers":33,"ai_summary":"Docetl is an agentic tool that leverages large language models to process unstructured data, perform ETL operations, and analyze documents.","readme_excerpt":"## Install\n\n```bash\npip install docetl\nexport OPENAI_API_KEY=your_key   # or any LLM provider key\n```\n\n---","github_created_at":"2024-07-09T05:57:16+00:00","created_at":"2026-07-15T10:47:35.619661+00:00","updated_at":"2026-09-20T04:27:15.889072+00:00","categories":[{"slug":"ai-agents","name":"AI Agents","url":"https://www.graphcanon.com/categories/ai-agents","markdown_url":"https://www.graphcanon.com/categories/ai-agents.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/ai-agents"},{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"}],"tags":[{"slug":"agents","name":"agents"},{"slug":"data","name":"data"},{"slug":"document-analysis","name":"document-analysis"},{"slug":"etl","name":"etl"},{"slug":"llm","name":"llm"},{"slug":"unstructured-data","name":"unstructured-data"}],"trust":{"provenance":{"is_fork":false,"github_id":826111692,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-09-15T06:00:57.552Z","maintenance":{"label":"Active","score":82,"methodology":"github_public_v1","releases_90d":0,"days_since_push":9,"last_release_at":"2026-06-17T03:25:51Z","stars_delta_30d":131,"open_issues_delta_30d":3},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-15T10:47:37.149Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-09-15T06:00:57.990Z"},"deploy":{"source":"dockerfile:Dockerfile","self_host":true,"observed_at":"2026-09-15T06:00:57.990Z","managed_saas":false},"has_cli":{"value":true,"source":"pyproject.toml:[project.scripts]","observed_at":"2026-09-15T06:00:57.990Z"},"languages":{"value":["python"],"source":"github.language+pyproject.toml","observed_at":"2026-09-15T06:00:57.990Z"},"has_docker":{"value":true,"source":"dockerfile:Dockerfile","observed_at":"2026-09-15T06:00:57.990Z"},"license_spdx":{"value":"MIT","source":"github.license","observed_at":"2026-09-15T06:00:57.990Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":{"min_ram_gb":null,"requires_docker":false},"constraints":{"min_ram_gb":null,"requires_docker":false},"when_to_use":["When you require integration with any LLM provider through API keys like OPENAI_API_KEY.","For tasks that involve complex parsing and analysis of unstructured data across multiple documents."],"when_not_to_use":["If your project strictly requires low-latency processing for real-time applications, as Docetl's agentic approach might introduce higher latency due to backend API calls.","In scenarios where the document datasets are predominantly structured or semi-structured, making traditional ETL tools more efficient."],"source":"enrich:decision_facts","observed_at":"2026-07-16T21:29:04.426Z"},"constraint_facets":{"min_ram_gb":null,"requires_docker":false},"decision_summary":[{"label":"Adopt for","value":"Docetl is an agentic system that employs large language models for data processing and ETL operations, specifically suited to handle unstructured document analysis tasks."}]}}