{"data":{"node":{"slug":"zipstack-unstract","name":"unstract","tagline":"LLM-Driven Extraction of Unstructured Data for API Deployments and ETL Pipeline Workflows","github_url":"https://github.com/Zipstack/unstract","owner":"Zipstack","repo":"unstract","owner_avatar_url":"https://avatars.githubusercontent.com/u/89070934?v=4","primary_language":"Python","stars":6932,"forks":663,"topics":["ai-agents","data-engineering","document-ai","generative-ai","idp","json-extraction","llm","mcp-server","ocr","pdf-extraction","prompt-engineering","structured-output"],"archived":false,"github_pushed_at":"2026-07-27T22:23:41+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/zipstack-unstract","markdown_url":"https://www.graphcanon.com/tools/zipstack-unstract.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zipstack-unstract","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zipstack-unstract"},"categories":[{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"},{"slug":"llm-frameworks","name":"LLM Frameworks","url":"https://www.graphcanon.com/categories/llm-frameworks","markdown_url":"https://www.graphcanon.com/categories/llm-frameworks.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/llm-frameworks"}],"tags":[{"slug":"ai-agents","name":"ai-agents"},{"slug":"data-engineering","name":"data-engineering"},{"slug":"document-ai","name":"document-ai"},{"slug":"generative-ai","name":"generative-ai"},{"slug":"idp","name":"idp"},{"slug":"json-extraction","name":"json-extraction"},{"slug":"llm","name":"llm"},{"slug":"ocr","name":"ocr"}],"edges":[],"neighbours":[{"slug":"graphify-labs-graphify","name":"graphify","tagline":"Turn any code or documentation into a queryable knowledge graph","github_url":"https://github.com/Graphify-Labs/graphify","owner":"Graphify-Labs","repo":"graphify","owner_avatar_url":"https://avatars.githubusercontent.com/u/297659074?v=4","primary_language":"Python","stars":107507,"forks":10441,"topics":["ai-agents","antigravity","ast","claude-code","code-analysis","code-search","codex","cursor","developer-tools","gemini","graphrag","knowledge-graph","leiden","llm","mcp","openclaw","rag","skills","tree-sitter"],"archived":false,"github_pushed_at":"2026-08-17T18:42:58+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/graphify-labs-graphify","markdown_url":"https://www.graphcanon.com/tools/graphify-labs-graphify.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/graphify-labs-graphify","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=graphify-labs-graphify","shared_categories":["data-retrieval"]},{"slug":"pathwaycom-llm-app","name":"llm-app","tagline":"Ready-to-run cloud templates for RAG, AI pipelines, and enterprise search with live data.","github_url":"https://github.com/pathwaycom/llm-app","owner":"pathwaycom","repo":"llm-app","owner_avatar_url":"https://avatars.githubusercontent.com/u/25750857?v=4","primary_language":"Jupyter Notebook","stars":59037,"forks":1466,"topics":["chatbot","hugging-face","llm","llm-local","llm-prompting","llm-security","llmops","machine-learning","open-ai","pathway","rag","real-time","retrieval-augmented-generation","vector-database","vector-index"],"archived":false,"github_pushed_at":"2026-07-05T17:59:07+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/pathwaycom-llm-app","markdown_url":"https://www.graphcanon.com/tools/pathwaycom-llm-app.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pathwaycom-llm-app","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pathwaycom-llm-app","shared_categories":["data-retrieval","llm-frameworks"]},{"slug":"scrapegraphai-scrapegraph-ai","name":"Scrapegraph-ai","tagline":"Python scraper based on AI","github_url":"https://github.com/ScrapeGraphAI/Scrapegraph-ai","owner":"ScrapeGraphAI","repo":"Scrapegraph-ai","owner_avatar_url":"https://avatars.githubusercontent.com/u/171017415?v=4","primary_language":"Python","stars":29618,"forks":2925,"topics":["ai-crawler","ai-scraping","ai-search","crawler","data-extraction","firecrawl-alternative","large-language-model","llm","markdown","rag","scraping","scraping-python","web-crawler","web-crawlers","web-data","web-data-extraction","web-scraper","web-scraping","web-search","webscraping"],"archived":false,"github_pushed_at":"2026-07-20T14:22:20+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/scrapegraphai-scrapegraph-ai","markdown_url":"https://www.graphcanon.com/tools/scrapegraphai-scrapegraph-ai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/scrapegraphai-scrapegraph-ai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=scrapegraphai-scrapegraph-ai","shared_categories":["data-retrieval","llm-frameworks"]},{"slug":"opendataloader-project-opendataloader-pdf","name":"opendataloader-pdf","tagline":"PDF Parser for AI-ready data","github_url":"https://github.com/opendataloader-project/opendataloader-pdf","owner":"opendataloader-project","repo":"opendataloader-pdf","owner_avatar_url":"https://avatars.githubusercontent.com/u/211280852?v=4","primary_language":"Java","stars":28528,"forks":2724,"topics":["a11y","accessibility","ai","bounding-box","document-parsing","eaa","html","json","markdown","ocr","ocr-recognition","pdf","pdf-accessibility","pdf-converter","pdf-extraction","pdf-parser","pdf-ua","rag","tables","tagged-pdf"],"archived":false,"github_pushed_at":"2026-08-18T04:01:45+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/opendataloader-project-opendataloader-pdf","markdown_url":"https://www.graphcanon.com/tools/opendataloader-project-opendataloader-pdf.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/opendataloader-project-opendataloader-pdf","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=opendataloader-project-opendataloader-pdf","shared_categories":["data-retrieval"]},{"slug":"promtengineer-localgpt","name":"localGPT","tagline":"Chat with your documents locally using GPT models","github_url":"https://github.com/PromtEngineer/localGPT","owner":"PromtEngineer","repo":"localGPT","owner_avatar_url":"https://avatars.githubusercontent.com/u/134474669?v=4","primary_language":"Python","stars":22209,"forks":2467,"topics":[],"archived":false,"github_pushed_at":"2026-07-18T07:15:01+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/promtengineer-localgpt","markdown_url":"https://www.graphcanon.com/tools/promtengineer-localgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/promtengineer-localgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=promtengineer-localgpt","shared_categories":["data-retrieval"]},{"slug":"tencent-weknora","name":"WeKnora","tagline":"Open-source LLM knowledge platform for creating a queryable RAG, autonomous reasoning agent, and self-maintaining Wiki.","github_url":"https://github.com/Tencent/WeKnora","owner":"Tencent","repo":"WeKnora","owner_avatar_url":"https://avatars.githubusercontent.com/u/18461506?v=4","primary_language":"Go","stars":19992,"forks":2877,"topics":["agent","agentic","ai","chatbot","embeddings","evaluation","generative-ai","golang","knowledge-base","llm","multi-tenant","multimodel","ollama","openai","question-answering","rag","reranking","semantic-search","vector-search","wiki"],"archived":false,"github_pushed_at":"2026-08-16T05:31:52+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/tencent-weknora","markdown_url":"https://www.graphcanon.com/tools/tencent-weknora.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tencent-weknora","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tencent-weknora","shared_categories":["llm-frameworks"]},{"slug":"unstructured-io-unstructured","name":"unstructured","tagline":"Convert documents to structured data effortlessly","github_url":"https://github.com/Unstructured-IO/unstructured","owner":"Unstructured-IO","repo":"unstructured","owner_avatar_url":"https://avatars.githubusercontent.com/u/108372208?v=4","primary_language":"HTML","stars":15238,"forks":1284,"topics":["data-pipelines","deep-learning","document-image-analysis","document-image-processing","document-parser","document-parsing","docx","donut","information-retrieval","langchain","llm","machine-learning","ml","natural-language-processing","nlp","ocr","pdf","pdf-to-json","pdf-to-text","preprocessing"],"archived":false,"github_pushed_at":"2026-07-31T20:54:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/unstructured-io-unstructured","markdown_url":"https://www.graphcanon.com/tools/unstructured-io-unstructured.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/unstructured-io-unstructured","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=unstructured-io-unstructured","shared_categories":["data-retrieval"]},{"slug":"yusufkaraaslan-skill-seekers","name":"Skill_Seekers","tagline":"Automation tool for converting documentation and code into Claude AI skills","github_url":"https://github.com/yusufkaraaslan/Skill_Seekers","owner":"yusufkaraaslan","repo":"Skill_Seekers","owner_avatar_url":"https://avatars.githubusercontent.com/u/11597362?v=4","primary_language":"Python","stars":14827,"forks":1513,"topics":["ai-tools","ast-parser","automation","claude-ai","claude-skills","code-analysis","conflict-detection","documentation","documentation-generator","github","github-scraper","mcp","mcp-server","multi-source","ocr","pdf","python","web-scraping"],"archived":false,"github_pushed_at":"2026-08-25T04:24:33+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/yusufkaraaslan-skill-seekers","markdown_url":"https://www.graphcanon.com/tools/yusufkaraaslan-skill-seekers.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/yusufkaraaslan-skill-seekers","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=yusufkaraaslan-skill-seekers","shared_categories":[]},{"slug":"data-privacy-stack-presidio","name":"presidio","tagline":"A framework for detecting and anonymizing sensitive data","github_url":"https://github.com/data-privacy-stack/presidio","owner":"data-privacy-stack","repo":"presidio","owner_avatar_url":"https://avatars.githubusercontent.com/u/275623515?v=4","primary_language":"Python","stars":10395,"forks":1237,"topics":["anonymization","data-anonymization","data-masking","data-obfuscation","data-privacy","data-redaction","de-identification","guardrails","image-redactor","named-entity-recognition","nlp","personally-identifiable-information","phi","pii","pii-detection","privacy","python","sensitive-data","spacy","transformers"],"archived":false,"github_pushed_at":"2026-08-08T21:25:09+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/data-privacy-stack-presidio","markdown_url":"https://www.graphcanon.com/tools/data-privacy-stack-presidio.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/data-privacy-stack-presidio","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=data-privacy-stack-presidio","shared_categories":["data-retrieval"]},{"slug":"datajuicer-data-juicer","name":"data-juicer","tagline":"Data processing for and with foundation models","github_url":"https://github.com/datajuicer/data-juicer","owner":"datajuicer","repo":"data-juicer","owner_avatar_url":"https://avatars.githubusercontent.com/u/223222708?v=4","primary_language":"Python","stars":6897,"forks":404,"topics":["data","data-analysis","data-pipeline","data-processing","data-science","data-visualization","foundation-models","instruction-tuning","large-language-models","llm","llms","multi-modal","pre-training","synthetic-data"],"archived":false,"github_pushed_at":"2026-08-13T09:19:31+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/datajuicer-data-juicer","markdown_url":"https://www.graphcanon.com/tools/datajuicer-data-juicer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/datajuicer-data-juicer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=datajuicer-data-juicer","shared_categories":["data-retrieval"]},{"slug":"clusterzx-paperless-ai","name":"paperless-ai","tagline":"Automated document analyzer for Paperless-ngx using OpenAI API and compatible services to tag documents","github_url":"https://github.com/clusterzx/paperless-ai","owner":"clusterzx","repo":"paperless-ai","owner_avatar_url":"https://avatars.githubusercontent.com/u/32274973?v=4","primary_language":"JavaScript","stars":5882,"forks":322,"topics":["ai","automation","gemma","llama","mistral","ollama","paperless","paperless-ngx","phi"],"archived":false,"github_pushed_at":"2026-08-02T04:51:37+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/clusterzx-paperless-ai","markdown_url":"https://www.graphcanon.com/tools/clusterzx-paperless-ai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/clusterzx-paperless-ai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=clusterzx-paperless-ai","shared_categories":[]},{"slug":"ucbepic-docetl","name":"docetl","tagline":"A system for agentic LLM-powered data processing and ETL","github_url":"https://github.com/ucbepic/docetl","owner":"ucbepic","repo":"docetl","owner_avatar_url":"https://avatars.githubusercontent.com/u/88680502?v=4","primary_language":"Python","stars":3961,"forks":421,"topics":["agents","data","data-pipelines","document-analysis","document-processing","elt","etl","llm","python","semantic-data","unstructured-data","unstructured-data-analysis","workflow"],"archived":false,"github_pushed_at":"2026-08-09T23:31:04+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ucbepic-docetl","markdown_url":"https://www.graphcanon.com/tools/ucbepic-docetl.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ucbepic-docetl","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ucbepic-docetl","shared_categories":["data-retrieval"]}]}}