{"data":{"node":{"slug":"ucbepic-docetl","name":"docetl","tagline":"A system for agentic LLM-powered data processing and ETL","github_url":"https://github.com/ucbepic/docetl","owner":"ucbepic","repo":"docetl","owner_avatar_url":"https://avatars.githubusercontent.com/u/88680502?v=4","primary_language":"Python","stars":4092,"forks":443,"topics":["agents","data","data-pipelines","document-analysis","document-processing","elt","etl","llm","python","semantic-data","unstructured-data","unstructured-data-analysis","workflow"],"archived":false,"github_pushed_at":"2026-09-05T16:55:18+00:00","maintenance_label":"Active","stars_delta_30d":131,"url":"https://www.graphcanon.com/tools/ucbepic-docetl","markdown_url":"https://www.graphcanon.com/tools/ucbepic-docetl.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ucbepic-docetl","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ucbepic-docetl"},"categories":[{"slug":"ai-agents","name":"AI Agents","url":"https://www.graphcanon.com/categories/ai-agents","markdown_url":"https://www.graphcanon.com/categories/ai-agents.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/ai-agents"},{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"}],"tags":[{"slug":"agents","name":"agents"},{"slug":"data","name":"data"},{"slug":"document-analysis","name":"document-analysis"},{"slug":"etl","name":"etl"},{"slug":"llm","name":"llm"},{"slug":"unstructured-data","name":"unstructured-data"}],"edges":[],"neighbours":[{"slug":"pathwaycom-llm-app","name":"llm-app","tagline":"Ready-to-run cloud templates for RAG, AI pipelines, and enterprise search with live data","github_url":"https://github.com/pathwaycom/llm-app","owner":"pathwaycom","repo":"llm-app","owner_avatar_url":"https://avatars.githubusercontent.com/u/25750857?v=4","primary_language":"Jupyter Notebook","stars":58920,"forks":1498,"topics":["chatbot","hugging-face","llm","llm-local","llm-prompting","llm-security","llmops","machine-learning","open-ai","pathway","rag","real-time","retrieval-augmented-generation","vector-database","vector-index"],"archived":false,"github_pushed_at":"2026-07-05T17:59:07+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/pathwaycom-llm-app","markdown_url":"https://www.graphcanon.com/tools/pathwaycom-llm-app.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pathwaycom-llm-app","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pathwaycom-llm-app","shared_categories":["data-retrieval"]},{"slug":"google-langextract","name":"langextract","tagline":"Python library for extracting structured information from unstructured text using LLMs","github_url":"https://github.com/google/langextract","owner":"google","repo":"langextract","owner_avatar_url":"https://avatars.githubusercontent.com/u/1342004?v=4","primary_language":"Python","stars":38614,"forks":2704,"topics":["gemini","gemini-ai","gemini-api","gemini-flash","gemini-pro","information-extration","large-language-models","llm","nlp","python","structured-data"],"archived":false,"github_pushed_at":"2026-09-13T18:57:36+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/google-langextract","markdown_url":"https://www.graphcanon.com/tools/google-langextract.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/google-langextract","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=google-langextract","shared_categories":["data-retrieval"]},{"slug":"nirdiamant-agents-towards-production","name":"agents-towards-production","tagline":"End-to-end, code-first tutorials for building production-grade GenAI agents","github_url":"https://github.com/NirDiamant/agents-towards-production","owner":"NirDiamant","repo":"agents-towards-production","owner_avatar_url":"https://avatars.githubusercontent.com/u/28316913?v=4","primary_language":"Jupyter Notebook","stars":21298,"forks":2824,"topics":["agent","agent-framework","agentic-ai","agents","ai-agents","deployment","genai","generative-ai","langgraph","llm","llms","mcp","mlops","multi-agent-systems","observability","production","python","rag","tutorials"],"archived":false,"github_pushed_at":"2026-08-15T00:52:10+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/nirdiamant-agents-towards-production","markdown_url":"https://www.graphcanon.com/tools/nirdiamant-agents-towards-production.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nirdiamant-agents-towards-production","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nirdiamant-agents-towards-production","shared_categories":["ai-agents"]},{"slug":"tencent-weknora","name":"WeKnora","tagline":"Open-source LLM knowledge platform for creating a queryable RAG, autonomous reasoning agent, and self-maintaining Wiki.","github_url":"https://github.com/Tencent/WeKnora","owner":"Tencent","repo":"WeKnora","owner_avatar_url":"https://avatars.githubusercontent.com/u/18461506?v=4","primary_language":"Go","stars":19992,"forks":2877,"topics":["agent","agentic","ai","chatbot","embeddings","evaluation","generative-ai","golang","knowledge-base","llm","multi-tenant","multimodel","ollama","openai","question-answering","rag","reranking","semantic-search","vector-search","wiki"],"archived":false,"github_pushed_at":"2026-08-16T05:31:52+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/tencent-weknora","markdown_url":"https://www.graphcanon.com/tools/tencent-weknora.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tencent-weknora","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tencent-weknora","shared_categories":["ai-agents"]},{"slug":"unstructured-io-unstructured","name":"unstructured","tagline":"Convert documents to structured data effortlessly","github_url":"https://github.com/Unstructured-IO/unstructured","owner":"Unstructured-IO","repo":"unstructured","owner_avatar_url":"https://avatars.githubusercontent.com/u/108372208?v=4","primary_language":"HTML","stars":15446,"forks":1330,"topics":["data-pipelines","deep-learning","document-image-analysis","document-image-processing","document-parser","document-parsing","docx","donut","information-retrieval","langchain","llm","machine-learning","ml","natural-language-processing","nlp","ocr","pdf","pdf-to-json","pdf-to-text","preprocessing"],"archived":false,"github_pushed_at":"2026-09-15T00:40:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/unstructured-io-unstructured","markdown_url":"https://www.graphcanon.com/tools/unstructured-io-unstructured.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/unstructured-io-unstructured","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=unstructured-io-unstructured","shared_categories":["data-retrieval"]},{"slug":"databendlabs-databend","name":"databend","tagline":"All-in-One Data Warehouse: Analytics, Search, AI, and Python Sandboxing Reimagined From Scratch.","github_url":"https://github.com/databendlabs/databend","owner":"databendlabs","repo":"databend","owner_avatar_url":"https://avatars.githubusercontent.com/u/80994548?v=4","primary_language":"Rust","stars":9444,"forks":897,"topics":["ai","bigdata","cloud-native","database","elasticsearch","geospatial","lakehouse","olap","rust","serverless","snowflake","sql","vector-database","vector-search"],"archived":false,"github_pushed_at":"2026-09-20T05:38:04+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/databendlabs-databend","markdown_url":"https://www.graphcanon.com/tools/databendlabs-databend.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/databendlabs-databend","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=databendlabs-databend","shared_categories":["data-retrieval"]},{"slug":"mage-ai-mage-ai","name":"mage-ai","tagline":"Build, run and manage data pipelines for integrating and transforming data","github_url":"https://github.com/mage-ai/mage-ai","owner":"mage-ai","repo":"mage-ai","owner_avatar_url":"https://avatars.githubusercontent.com/u/69371472?v=4","primary_language":"Python","stars":8823,"forks":990,"topics":["artificial-intelligence","data","data-engineering","data-integration","data-pipelines","data-science","dbt","elt","etl","machine-learning","orchestration","pipeline","pipelines","python","reverse-etl","spark","sql","transformation"],"archived":false,"github_pushed_at":"2026-09-11T19:20:35+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/mage-ai-mage-ai","markdown_url":"https://www.graphcanon.com/tools/mage-ai-mage-ai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/mage-ai-mage-ai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=mage-ai-mage-ai","shared_categories":["data-retrieval"]},{"slug":"zipstack-unstract","name":"unstract","tagline":"LLM-Driven Extraction of Unstructured Data for API Deployments and ETL Pipeline Workflows","github_url":"https://github.com/Zipstack/unstract","owner":"Zipstack","repo":"unstract","owner_avatar_url":"https://avatars.githubusercontent.com/u/89070934?v=4","primary_language":"Python","stars":7245,"forks":718,"topics":["ai-agents","data-engineering","document-ai","generative-ai","idp","json-extraction","llm","mcp-server","ocr","pdf-extraction","prompt-engineering","structured-output"],"archived":false,"github_pushed_at":"2026-09-18T12:40:20+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/zipstack-unstract","markdown_url":"https://www.graphcanon.com/tools/zipstack-unstract.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zipstack-unstract","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zipstack-unstract","shared_categories":["data-retrieval"]},{"slug":"datajuicer-data-juicer","name":"data-juicer","tagline":"Data processing for and with foundation models","github_url":"https://github.com/datajuicer/data-juicer","owner":"datajuicer","repo":"data-juicer","owner_avatar_url":"https://avatars.githubusercontent.com/u/223222708?v=4","primary_language":"Python","stars":6897,"forks":404,"topics":["data","data-analysis","data-pipeline","data-processing","data-science","data-visualization","foundation-models","instruction-tuning","large-language-models","llm","llms","multi-modal","pre-training","synthetic-data"],"archived":false,"github_pushed_at":"2026-08-13T09:19:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/datajuicer-data-juicer","markdown_url":"https://www.graphcanon.com/tools/datajuicer-data-juicer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/datajuicer-data-juicer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=datajuicer-data-juicer","shared_categories":["data-retrieval"]},{"slug":"giovannipasq-agentic-rag-for-dummies","name":"agentic-rag-for-dummies","tagline":"A modular Agentic RAG built with LangGraph for learning Retrieval-Augmented Generation Agents","github_url":"https://github.com/GiovanniPasq/agentic-rag-for-dummies","owner":"GiovanniPasq","repo":"agentic-rag-for-dummies","owner_avatar_url":"https://avatars.githubusercontent.com/u/33225259?v=4","primary_language":"Jupyter Notebook","stars":4188,"forks":552,"topics":["agent","agentic-ai","agentic-rag","agents","ai-agents","bm25","generative-ai","gradio","langchain","langgraph","llm","ollama","qdrant","rag","rag-agents","rag-chatbot","rag-pipeline","retrieval-augmented-generation","retrieval-augmented-generation-rag"],"archived":false,"github_pushed_at":"2026-08-30T10:19:02+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/giovannipasq-agentic-rag-for-dummies","markdown_url":"https://www.graphcanon.com/tools/giovannipasq-agentic-rag-for-dummies.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/giovannipasq-agentic-rag-for-dummies","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=giovannipasq-agentic-rag-for-dummies","shared_categories":["ai-agents","data-retrieval"]},{"slug":"huggingface-datatrove","name":"datatrove","tagline":"Platform-agnostic customizable pipeline processing blocks for data processing and transformation.","github_url":"https://github.com/huggingface/datatrove","owner":"huggingface","repo":"datatrove","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":3324,"forks":297,"topics":[],"archived":false,"github_pushed_at":"2026-08-13T16:04:51+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huggingface-datatrove","markdown_url":"https://www.graphcanon.com/tools/huggingface-datatrove.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-datatrove","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-datatrove","shared_categories":["data-retrieval"]},{"slug":"spiceai-spiceai","name":"spiceai","tagline":"A real-time analytics node for data-grounded AI applications","github_url":"https://github.com/spiceai/spiceai","owner":"spiceai","repo":"spiceai","owner_avatar_url":"https://avatars.githubusercontent.com/u/73862742?v=4","primary_language":"Rust","stars":3084,"forks":230,"topics":["artificial-intelligence","data","data-federation","developers","full-text-search","infrastructure","llm-inference","machine-learning","sql"],"archived":false,"github_pushed_at":"2026-09-19T18:28:38+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/spiceai-spiceai","markdown_url":"https://www.graphcanon.com/tools/spiceai-spiceai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/spiceai-spiceai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=spiceai-spiceai","shared_categories":["data-retrieval"]}]}}