{"data":{"node":{"slug":"data-privacy-stack-presidio","name":"presidio","tagline":"A framework for detecting and anonymizing sensitive data","github_url":"https://github.com/data-privacy-stack/presidio","owner":"data-privacy-stack","repo":"presidio","owner_avatar_url":"https://avatars.githubusercontent.com/u/275623515?v=4","primary_language":"Python","stars":10818,"forks":1279,"topics":["anonymization","data-anonymization","data-masking","data-obfuscation","data-privacy","data-redaction","de-identification","guardrails","image-redactor","named-entity-recognition","nlp","personally-identifiable-information","phi","pii","pii-detection","privacy","python","sensitive-data","spacy","transformers"],"archived":false,"github_pushed_at":"2026-09-10T15:20:43+00:00","maintenance_label":"Very active","stars_delta_30d":423,"url":"https://www.graphcanon.com/tools/data-privacy-stack-presidio","markdown_url":"https://www.graphcanon.com/tools/data-privacy-stack-presidio.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/data-privacy-stack-presidio","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=data-privacy-stack-presidio"},"categories":[{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"},{"slug":"evaluation-observability","name":"Evaluation & Observability","url":"https://www.graphcanon.com/categories/evaluation-observability","markdown_url":"https://www.graphcanon.com/categories/evaluation-observability.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/evaluation-observability"}],"tags":[{"slug":"data-anonymization","name":"data-anonymization"},{"slug":"data-obfuscation","name":"data-obfuscation"},{"slug":"group-python-frameworks","name":"group:python-frameworks"}],"edges":[],"neighbours":[{"slug":"pathwaycom-llm-app","name":"llm-app","tagline":"Ready-to-run cloud templates for RAG, AI pipelines, and enterprise search with live data","github_url":"https://github.com/pathwaycom/llm-app","owner":"pathwaycom","repo":"llm-app","owner_avatar_url":"https://avatars.githubusercontent.com/u/25750857?v=4","primary_language":"Jupyter Notebook","stars":58920,"forks":1498,"topics":["chatbot","hugging-face","llm","llm-local","llm-prompting","llm-security","llmops","machine-learning","open-ai","pathway","rag","real-time","retrieval-augmented-generation","vector-database","vector-index"],"archived":false,"github_pushed_at":"2026-07-05T17:59:07+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/pathwaycom-llm-app","markdown_url":"https://www.graphcanon.com/tools/pathwaycom-llm-app.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pathwaycom-llm-app","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pathwaycom-llm-app","shared_categories":["evaluation-observability","data-retrieval"]},{"slug":"google-langextract","name":"langextract","tagline":"Python library for extracting structured information from unstructured text using LLMs","github_url":"https://github.com/google/langextract","owner":"google","repo":"langextract","owner_avatar_url":"https://avatars.githubusercontent.com/u/1342004?v=4","primary_language":"Python","stars":38614,"forks":2704,"topics":["gemini","gemini-ai","gemini-api","gemini-flash","gemini-pro","information-extration","large-language-models","llm","nlp","python","structured-data"],"archived":false,"github_pushed_at":"2026-09-13T18:57:36+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/google-langextract","markdown_url":"https://www.graphcanon.com/tools/google-langextract.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/google-langextract","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=google-langextract","shared_categories":["data-retrieval"]},{"slug":"opendataloader-project-opendataloader-pdf","name":"opendataloader-pdf","tagline":"PDF Parser for AI-ready data","github_url":"https://github.com/opendataloader-project/opendataloader-pdf","owner":"opendataloader-project","repo":"opendataloader-pdf","owner_avatar_url":"https://avatars.githubusercontent.com/u/211280852?v=4","primary_language":"Java","stars":28528,"forks":2724,"topics":["a11y","accessibility","ai","bounding-box","document-parsing","eaa","html","json","markdown","ocr","ocr-recognition","pdf","pdf-accessibility","pdf-converter","pdf-extraction","pdf-parser","pdf-ua","rag","tables","tagged-pdf"],"archived":false,"github_pushed_at":"2026-08-18T04:01:45+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/opendataloader-project-opendataloader-pdf","markdown_url":"https://www.graphcanon.com/tools/opendataloader-project-opendataloader-pdf.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/opendataloader-project-opendataloader-pdf","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=opendataloader-project-opendataloader-pdf","shared_categories":["data-retrieval"]},{"slug":"promtengineer-localgpt","name":"localGPT","tagline":"Chat with your documents locally using GPT models","github_url":"https://github.com/PromtEngineer/localGPT","owner":"PromtEngineer","repo":"localGPT","owner_avatar_url":"https://avatars.githubusercontent.com/u/134474669?v=4","primary_language":"Python","stars":22202,"forks":2461,"topics":[],"archived":false,"github_pushed_at":"2026-08-26T04:52:30+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/promtengineer-localgpt","markdown_url":"https://www.graphcanon.com/tools/promtengineer-localgpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/promtengineer-localgpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=promtengineer-localgpt","shared_categories":["data-retrieval"]},{"slug":"unstructured-io-unstructured","name":"unstructured","tagline":"Convert documents to structured data effortlessly","github_url":"https://github.com/Unstructured-IO/unstructured","owner":"Unstructured-IO","repo":"unstructured","owner_avatar_url":"https://avatars.githubusercontent.com/u/108372208?v=4","primary_language":"HTML","stars":15446,"forks":1330,"topics":["data-pipelines","deep-learning","document-image-analysis","document-image-processing","document-parser","document-parsing","docx","donut","information-retrieval","langchain","llm","machine-learning","ml","natural-language-processing","nlp","ocr","pdf","pdf-to-json","pdf-to-text","preprocessing"],"archived":false,"github_pushed_at":"2026-09-15T00:40:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/unstructured-io-unstructured","markdown_url":"https://www.graphcanon.com/tools/unstructured-io-unstructured.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/unstructured-io-unstructured","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=unstructured-io-unstructured","shared_categories":["data-retrieval"]},{"slug":"zilliztech-deep-searcher","name":"deep-searcher","tagline":"Open Source Deep Research Alternative to Reason and Search on Private Data.","github_url":"https://github.com/zilliztech/deep-searcher","owner":"zilliztech","repo":"deep-searcher","owner_avatar_url":"https://avatars.githubusercontent.com/u/18416694?v=4","primary_language":"Python","stars":8060,"forks":775,"topics":["agent","agentic-rag","claude","deep-research","deepseek","deepseek-r1","grok","grok3","llama4","llm","milvus","openai","qwen3","rag","reasoning-models","vector-database","zilliz"],"archived":false,"github_pushed_at":"2025-11-19T06:04:16+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/zilliztech-deep-searcher","markdown_url":"https://www.graphcanon.com/tools/zilliztech-deep-searcher.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zilliztech-deep-searcher","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zilliztech-deep-searcher","shared_categories":[]},{"slug":"zipstack-unstract","name":"unstract","tagline":"LLM-Driven Extraction of Unstructured Data for API Deployments and ETL Pipeline Workflows","github_url":"https://github.com/Zipstack/unstract","owner":"Zipstack","repo":"unstract","owner_avatar_url":"https://avatars.githubusercontent.com/u/89070934?v=4","primary_language":"Python","stars":7245,"forks":718,"topics":["ai-agents","data-engineering","document-ai","generative-ai","idp","json-extraction","llm","mcp-server","ocr","pdf-extraction","prompt-engineering","structured-output"],"archived":false,"github_pushed_at":"2026-09-18T12:40:20+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/zipstack-unstract","markdown_url":"https://www.graphcanon.com/tools/zipstack-unstract.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zipstack-unstract","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zipstack-unstract","shared_categories":["data-retrieval"]},{"slug":"datajuicer-data-juicer","name":"data-juicer","tagline":"Data processing for and with foundation models","github_url":"https://github.com/datajuicer/data-juicer","owner":"datajuicer","repo":"data-juicer","owner_avatar_url":"https://avatars.githubusercontent.com/u/223222708?v=4","primary_language":"Python","stars":6897,"forks":404,"topics":["data","data-analysis","data-pipeline","data-processing","data-science","data-visualization","foundation-models","instruction-tuning","large-language-models","llm","llms","multi-modal","pre-training","synthetic-data"],"archived":false,"github_pushed_at":"2026-08-13T09:19:31+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/datajuicer-data-juicer","markdown_url":"https://www.graphcanon.com/tools/datajuicer-data-juicer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/datajuicer-data-juicer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=datajuicer-data-juicer","shared_categories":["data-retrieval"]},{"slug":"superagent-ai-superagent","name":"superagent","tagline":"Superagent SDK","github_url":"https://github.com/superagent-ai/superagent","owner":"superagent-ai","repo":"superagent","owner_avatar_url":"https://avatars.githubusercontent.com/u/152537519?v=4","primary_language":"TypeScript","stars":6752,"forks":964,"topics":["ai","anthropic","guardrails","llm","openai","prompt-injection","security"],"archived":false,"github_pushed_at":"2026-08-25T09:45:01+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/superagent-ai-superagent","markdown_url":"https://www.graphcanon.com/tools/superagent-ai-superagent.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/superagent-ai-superagent","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=superagent-ai-superagent","shared_categories":["evaluation-observability"]},{"slug":"maziyarpanahi-openmed","name":"openmed","tagline":"Local-first healthcare AI for clinical NER and HIPAA PII de-identification.","github_url":"https://github.com/maziyarpanahi/openmed","owner":"maziyarpanahi","repo":"openmed","owner_avatar_url":"https://avatars.githubusercontent.com/u/5762953?v=4","primary_language":"Python","stars":5346,"forks":680,"topics":["android","clinical-nlp","healthcare","hipaa","ios","javascript","llm","local-llm","mlx","ner","nlp","on-device","on-premise","pii","pii-detection","python","skills","sovereign-ai","swift","swiftui"],"archived":false,"github_pushed_at":"2026-09-19T13:42:26+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/maziyarpanahi-openmed","markdown_url":"https://www.graphcanon.com/tools/maziyarpanahi-openmed.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/maziyarpanahi-openmed","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=maziyarpanahi-openmed","shared_categories":[]},{"slug":"nyldn-claude-octopus","name":"claude-octopus","tagline":"Surface AI blindspots before you ship","github_url":"https://github.com/nyldn/claude-octopus","owner":"nyldn","repo":"claude-octopus","owner_avatar_url":"https://avatars.githubusercontent.com/u/4805949?v=4","primary_language":"Shell","stars":4090,"forks":378,"topics":["ai-agents","ai-orchestration","claude-code","claude-code-plugin","codex","copilot","developer-tools","double-diamond","gemini","multi-ai","multi-llm","ollama"],"archived":false,"github_pushed_at":"2026-09-20T05:07:44+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/nyldn-claude-octopus","markdown_url":"https://www.graphcanon.com/tools/nyldn-claude-octopus.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nyldn-claude-octopus","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nyldn-claude-octopus","shared_categories":[]},{"slug":"huggingface-datatrove","name":"datatrove","tagline":"Platform-agnostic customizable pipeline processing blocks for data processing and transformation.","github_url":"https://github.com/huggingface/datatrove","owner":"huggingface","repo":"datatrove","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":3324,"forks":297,"topics":[],"archived":false,"github_pushed_at":"2026-08-13T16:04:51+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huggingface-datatrove","markdown_url":"https://www.graphcanon.com/tools/huggingface-datatrove.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-datatrove","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-datatrove","shared_categories":["data-retrieval"]}]}}