{"data":{"slug":"google-langextract","name":"langextract","tagline":"A Python library for extracting structured information from unstructured text using LLMs.","github_url":"https://github.com/google/langextract","owner":"google","repo":"langextract","owner_avatar_url":"https://avatars.githubusercontent.com/u/1342004?v=4","primary_language":"Python","stars":38400,"forks":2693,"topics":["gemini","gemini-ai","gemini-api","gemini-flash","gemini-pro","information-extration","large-language-models","llm","nlp","python","structured-data"],"archived":false,"github_pushed_at":"2026-08-11T15:31:39+00:00","maintenance_label":"Very active","stars_delta_30d":1241,"url":"https://www.graphcanon.com/tools/google-langextract","markdown_url":"https://www.graphcanon.com/tools/google-langextract.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/google-langextract","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=google-langextract","description":"A Python library for extracting structured information from unstructured text using LLMs with precise source grounding and interactive visualization.","homepage_url":"https://pypi.org/project/langextract/","license":"Apache-2.0","open_issues":122,"watchers":169,"ai_summary":"google/langextract is a Python-based tool that uses large language models to extract and structure data from unstructured text sources, offering precise source grounding and interactive visualization features.","readme_excerpt":"## Quick Start\n\n> **Note:** Using cloud-hosted models like Gemini requires an API key. See the [API Key Setup](#api-key-setup-for-cloud-models) section for instructions on how to get and configure your key.\n\nExtract structured information with just a few lines of code.\n\n---\n\n# For basic installation:\npip install -e .\n\n---\n\n### Docker\n\n```bash\ndocker build -t langextract .\ndocker run --rm -e LANGEXTRACT_API_KEY=\"your-api-key\" langextract python your_script.py\n```\n\n---\n\n# Install with test dependencies\npip install -e \".[test]\"","github_created_at":"2025-07-08T20:46:06+00:00","created_at":"2026-07-07T17:31:42.949054+00:00","updated_at":"2026-08-16T12:01:53.851875+00:00","categories":[{"slug":"llm-frameworks","name":"LLM Frameworks","url":"https://www.graphcanon.com/categories/llm-frameworks","markdown_url":"https://www.graphcanon.com/categories/llm-frameworks.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/llm-frameworks"},{"slug":"model-training","name":"Model Training","url":"https://www.graphcanon.com/categories/model-training","markdown_url":"https://www.graphcanon.com/categories/model-training.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/model-training"}],"tags":[{"slug":"gemini","name":"gemini"},{"slug":"gemini-ai","name":"gemini-ai"},{"slug":"information-extraction","name":"information-extraction"},{"slug":"large-language-models","name":"large language models"},{"slug":"llm","name":"llm"},{"slug":"nlp","name":"nlp"},{"slug":"python","name":"python"},{"slug":"structured-data","name":"structured-data"}],"trust":{"provenance":{"is_fork":false,"github_id":1016323751,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-08-16T12:01:53.058Z","maintenance":{"label":"Very active","score":96,"methodology":"github_public_v1","releases_90d":2,"days_since_push":4,"last_release_at":"2026-07-02T06:23:27Z","stars_delta_30d":1241,"open_issues_delta_30d":15},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-11T10:57:43.599Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-08-16T12:01:53.523Z"},"deploy":{"source":"dockerfile:Dockerfile","self_host":true,"observed_at":"2026-08-16T12:01:53.523Z","managed_saas":false},"languages":{"value":["python"],"source":"github.language+pyproject.toml","observed_at":"2026-08-16T12:01:53.523Z"},"has_docker":{"value":true,"source":"dockerfile:Dockerfile","observed_at":"2026-08-16T12:01:53.523Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-08-16T12:01:53.523Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":null,"constraints":null,"when_to_use":["- When you require extraction of structured information with precise source references in your Python projects","- If you are working on NLP tasks that demand interactive visualization to better understand the extracted information"],"when_not_to_use":["- For tasks where real-time performance is critical, as langextract relies heavily on LLMs which may introduce latency","- When the project stack does not include Python or there's an existing strong preference for another programming language"],"source":"enrich:decision_facts","observed_at":"2026-07-11T12:46:31.673Z"},"constraint_facets":null,"decision_summary":[{"label":"Adopt for","value":"langextract is a Python library that leverages LLM capabilities to extract and structure data from unstructured text, providing features such as precise source grounding and interactive visualizations for improved data洞察"}]}}