{"data":{"slug":"adbar-trafilatura","name":"trafilatura","tagline":"Python & Command-line tool for web crawling, scraping and text extraction","github_url":"https://github.com/adbar/trafilatura","owner":"adbar","repo":"trafilatura","owner_avatar_url":"https://avatars.githubusercontent.com/u/2125866?v=4","primary_language":"Python","stars":6657,"forks":415,"topics":["article-extractor","corpus-builder","corpus-tools","crawler","html-to-markdown","html2text","llm","news-aggregator","news-crawler","nlp","rag","readability","rss-feed","scraping","tei","text-cleaning","text-extraction","text-mining","text-preprocessing","web-scraping"],"archived":false,"github_pushed_at":"2026-08-15T16:13:21+00:00","maintenance_label":"Very active","stars_delta_30d":343,"url":"https://www.graphcanon.com/tools/adbar-trafilatura","markdown_url":"https://www.graphcanon.com/tools/adbar-trafilatura.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/adbar-trafilatura","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=adbar-trafilatura","description":"Python & Command-line tool to gather text and metadata on the Web: Crawling, scraping, extraction, output as CSV, JSON, HTML, MD, TXT, XML","homepage_url":"https://trafilatura.readthedocs.io","license":"Apache-2.0","open_issues":62,"watchers":35,"ai_summary":"Provides functionalities for gathering text and metadata from the Web, including crawling, scraping, and various output formats.","readme_excerpt":"## License\n\nThis package is distributed under the [Apache 2.0 license](https://www.apache.org/licenses/LICENSE-2.0.html).\n\nVersions prior to v1.8.0 are under GPLv3+ license.","github_created_at":"2019-04-08T11:38:48+00:00","created_at":"2026-07-07T17:37:42.423006+00:00","updated_at":"2026-08-18T18:01:04.967186+00:00","categories":[{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"}],"tags":[{"slug":"article-extractor","name":"article-extractor"},{"slug":"corpus-builder","name":"corpus-builder"},{"slug":"text-mining","name":"text-mining"},{"slug":"web-scraping","name":"web-scraping"}],"trust":{"provenance":{"is_fork":false,"github_id":180136168,"owner_type":"User","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-08-18T18:01:04.148Z","maintenance":{"label":"Very active","score":96,"methodology":"github_public_v1","releases_90d":2,"days_since_push":3,"last_release_at":"2026-07-31T16:07:10Z","stars_delta_30d":343,"open_issues_delta_30d":-5},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-11T11:10:14.459Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-08-18T18:01:04.643Z"},"has_cli":{"value":true,"source":"pyproject.toml:[project.scripts]","observed_at":"2026-08-18T18:01:04.643Z"},"languages":{"value":["python"],"source":"github.language+pyproject.toml","observed_at":"2026-08-18T18:01:04.643Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-08-18T18:01:04.643Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":{"notes":["- Python environment installed on the system.","- Command line access required to use the command-line tool effectively."],"min_ram_gb":4,"requires_docker":false},"constraints":{"min_ram_gb":4,"requires_docker":false},"when_to_use":["- When you need both a programming interface and a command line tool to gather texts from the Web.","- To extract clean text directly in formats like CSV, JSON, HTML, Markdown, TXT, XML without needing to preprocess further."],"when_not_to_use":["- If your primary requirement is for real-time data streaming or highly interactive scraping tasks, as trafilatura provides robust but not real-time capabilities.","- For deep web or javascript-heavy sites where dynamic content rendering is necessary; it's more suited for static site and metadata extraction."],"source":"enrich:decision_facts","observed_at":"2026-07-12T09:40:17.652Z"},"constraint_facets":{"min_ram_gb":4,"requires_docker":false},"decision_summary":[{"label":"Requirements","value":"Min 4 GB RAM; - Python environment installed on the system.; - Command line access required to use the command-line tool effectively."},{"label":"Adopt for","value":"Trafilatura is a Python & command-line tool designed for web crawling, scraping, and text extraction."}]}}