{"data":{"node":{"slug":"datajuicer-data-juicer","name":"data-juicer","tagline":"Data processing for and with foundation models","github_url":"https://github.com/datajuicer/data-juicer","owner":"datajuicer","repo":"data-juicer","owner_avatar_url":"https://avatars.githubusercontent.com/u/223222708?v=4","primary_language":"Python","stars":6897,"forks":404,"topics":["data","data-analysis","data-pipeline","data-processing","data-science","data-visualization","foundation-models","instruction-tuning","large-language-models","llm","llms","multi-modal","pre-training","synthetic-data"],"archived":false,"github_pushed_at":"2026-08-13T09:19:31+00:00","maintenance_label":"Very active","stars_delta_30d":166,"url":"https://www.graphcanon.com/tools/datajuicer-data-juicer","markdown_url":"https://www.graphcanon.com/tools/datajuicer-data-juicer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/datajuicer-data-juicer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=datajuicer-data-juicer"},"categories":[{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"},{"slug":"model-training","name":"Model Training","url":"https://www.graphcanon.com/categories/model-training","markdown_url":"https://www.graphcanon.com/categories/model-training.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/model-training"}],"tags":[{"slug":"foundation-models","name":"foundation-models"},{"slug":"instruction-tuning","name":"instruction-tuning"},{"slug":"large-language-models","name":"large language models"},{"slug":"llm","name":"llm"},{"slug":"synthetic-data","name":"synthetic-data"}],"edges":[{"type":"depends_on","direction":"out","explanation":"Data-Juicer processes data which could include the output from unstructured documents processed by unstructured.","successor_context":null,"tool":{"slug":"unstructured-io-unstructured","name":"unstructured","tagline":"Convert documents to structured data effortlessly","github_url":"https://github.com/Unstructured-IO/unstructured","owner":"Unstructured-IO","repo":"unstructured","owner_avatar_url":"https://avatars.githubusercontent.com/u/108372208?v=4","primary_language":"HTML","stars":15238,"forks":1284,"topics":["data-pipelines","deep-learning","document-image-analysis","document-image-processing","document-parser","document-parsing","docx","donut","information-retrieval","langchain","llm","machine-learning","ml","natural-language-processing","nlp","ocr","pdf","pdf-to-json","pdf-to-text","preprocessing"],"archived":false,"github_pushed_at":"2026-07-31T20:54:17+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/unstructured-io-unstructured","markdown_url":"https://www.graphcanon.com/tools/unstructured-io-unstructured.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/unstructured-io-unstructured","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=unstructured-io-unstructured"}},{"type":"integrates_with","direction":"out","explanation":"ScrapeGraph-AI is focused on integrating AI for web scraping tasks, which could efficiently work with Data-Juicer's data processing capabilities to transform and analyze the scraped data.","successor_context":null,"tool":{"slug":"scrapegraphai-scrapegraph-ai","name":"Scrapegraph-ai","tagline":"Python scraper based on AI","github_url":"https://github.com/ScrapeGraphAI/Scrapegraph-ai","owner":"ScrapeGraphAI","repo":"Scrapegraph-ai","owner_avatar_url":"https://avatars.githubusercontent.com/u/171017415?v=4","primary_language":"Python","stars":29618,"forks":2925,"topics":["ai-crawler","ai-scraping","ai-search","crawler","data-extraction","firecrawl-alternative","large-language-model","llm","markdown","rag","scraping","scraping-python","web-crawler","web-crawlers","web-data","web-data-extraction","web-scraper","web-scraping","web-search","webscraping"],"archived":false,"github_pushed_at":"2026-07-20T14:22:20+00:00","maintenance_label":"Active","stars_delta_30d":1203,"url":"https://www.graphcanon.com/tools/scrapegraphai-scrapegraph-ai","markdown_url":"https://www.graphcanon.com/tools/scrapegraphai-scrapegraph-ai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/scrapegraphai-scrapegraph-ai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=scrapegraphai-scrapegraph-ai"}},{"type":"integrates_with","direction":"out","explanation":"Both Flowise and Data-Juicer aim at building AI workflows, where Flowise provides a visual interface for creating these, making it natural to potentially integrate data pipelines created or processed with Data-Juicer.","successor_context":null,"tool":{"slug":"flowiseai-flowise","name":"Flowise","tagline":"Build AI Agents, Visually","github_url":"https://github.com/FlowiseAI/Flowise","owner":"FlowiseAI","repo":"Flowise","owner_avatar_url":"https://avatars.githubusercontent.com/u/128289781?v=4","primary_language":"TypeScript","stars":55246,"forks":24869,"topics":["agentic-ai","agentic-workflow","agents","artificial-intelligence","chatbot","chatgpt","javascript","langchain","large-language-models","low-code","multiagent-systems","no-code","openai","rag","react","typescript","workflow-automation"],"archived":false,"github_pushed_at":"2026-08-07T05:55:02+00:00","maintenance_label":"Very active","stars_delta_30d":816,"url":"https://www.graphcanon.com/tools/flowiseai-flowise","markdown_url":"https://www.graphcanon.com/tools/flowiseai-flowise.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/flowiseai-flowise","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=flowiseai-flowise"}},{"type":"related","direction":"out","explanation":"Data-Juicer may generate datasets used in prompt creation and evaluation processes handled by PromptFoo.","successor_context":null,"tool":{"slug":"promptfoo-promptfoo","name":"promptfoo","tagline":"Tool for evaluating prompts and AI agents by comparing performance across various models and red teaming.","github_url":"https://github.com/promptfoo/promptfoo","owner":"promptfoo","repo":"promptfoo","owner_avatar_url":"https://avatars.githubusercontent.com/u/137907881?v=4","primary_language":"TypeScript","stars":23838,"forks":2147,"topics":["ci","ci-cd","cicd","evaluation","evaluation-framework","llm","llm-eval","llm-evaluation","llm-evaluation-framework","llmops","pentesting","prompt-engineering","prompt-testing","prompts","rag","red-teaming","testing","vulnerability-scanners"],"archived":false,"github_pushed_at":"2026-08-01T23:47:56+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/promptfoo-promptfoo","markdown_url":"https://www.graphcanon.com/tools/promptfoo-promptfoo.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/promptfoo-promptfoo","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=promptfoo-promptfoo"}},{"type":"integrates_with","direction":"out","explanation":"Data-Juicer processes raw data into AI-ready formats, while LangGraph orchestrates the deployment of stateful AI agents. The 'integrates with' relationship between them signifies that Data-Juicer's processed data can be seamlessly utilized by LangGraph for training or input to its orchestrated AI models, enabling a cohesive workflow from data preparation through to agent management.","successor_context":null,"tool":{"slug":"langchain-ai-langgraph","name":"langgraph","tagline":"Low-level orchestration framework for building stateful agents.","github_url":"https://github.com/langchain-ai/langgraph","owner":"langchain-ai","repo":"langgraph","owner_avatar_url":"https://avatars.githubusercontent.com/u/126733545?v=4","primary_language":"Python","stars":38352,"forks":6458,"topics":["agents","ai","ai-agents","chatgpt","deepagents","enterprise","framework","gemini","generative-ai","langchain","langgraph","llm","multiagent","open-source","openai","pydantic","python","rag"],"archived":false,"github_pushed_at":"2026-07-28T14:49:41+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/langchain-ai-langgraph","markdown_url":"https://www.graphcanon.com/tools/langchain-ai-langgraph.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/langchain-ai-langgraph","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=langchain-ai-langgraph"}},{"type":"integrates_with","direction":"out","explanation":"Both Kestra and Data-Juicer focus on data pipelines and processing, with Kestra being an orchestration platform for workflows that can include data processing tasks similar to those handled by Data-Juicer.","successor_context":null,"tool":{"slug":"kestra-io-kestra","name":"kestra","tagline":"Event Driven Orchestration & Scheduling Platform for Mission Critical Applications","github_url":"https://github.com/kestra-io/kestra","owner":"kestra-io","repo":"kestra","owner_avatar_url":"https://avatars.githubusercontent.com/u/59033362?v=4","primary_language":"Java","stars":27853,"forks":2922,"topics":["ai-agents","automation","control-plane","data-engineering","data-orchestration","data-orchestrator","devops","etl","hacktoberfest","high-availability","infra-automation","infra-ops","infrastructure-as-code","java","low-code","orchestration","pipeline","pipeline-as-code","scheduler","workflow"],"archived":false,"github_pushed_at":"2026-08-19T17:37:56+00:00","maintenance_label":"Very active","stars_delta_30d":455,"url":"https://www.graphcanon.com/tools/kestra-io-kestra","markdown_url":"https://www.graphcanon.com/tools/kestra-io-kestra.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/kestra-io-kestra","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=kestra-io-kestra"}},{"type":"integrates_with","direction":"out","explanation":"Data-juicer processes raw data into AI-ready formats, which can be integrated as input for Langchain-Chatchat's RAG system. Langchain-Chatchat uses this processed data to enhance its retrieval-augmented generation capabilities and agent applications.","successor_context":null,"tool":{"slug":"chatchat-space-langchain-chatchat","name":"Langchain-Chatchat","tagline":"Local knowledge-based RAG and Agent app using Langchain and various LLMs","github_url":"https://github.com/chatchat-space/Langchain-Chatchat","owner":"chatchat-space","repo":"Langchain-Chatchat","owner_avatar_url":"https://avatars.githubusercontent.com/u/139558948?v=4","primary_language":"Python","stars":38522,"forks":6266,"topics":["chatbot","chatchat","chatglm","chatgpt","embedding","faiss","fastchat","gpt","knowledge-base","langchain","langchain-chatglm","llama","llm","milvus","ollama","qwen","rag","retrieval-augmented-generation","streamlit","xinference"],"archived":false,"github_pushed_at":"2025-11-10T09:27:42+00:00","maintenance_label":"Slowing","stars_delta_30d":254,"url":"https://www.graphcanon.com/tools/chatchat-space-langchain-chatchat","markdown_url":"https://www.graphcanon.com/tools/chatchat-space-langchain-chatchat.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/chatchat-space-langchain-chatchat","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=chatchat-space-langchain-chatchat"}},{"type":"integrates_with","direction":"out","explanation":"Both Data-Juicer and pandas-ai involve data processing for AI tasks with specific emphasis on making it accessible via LLMs.","successor_context":null,"tool":{"slug":"sinaptik-ai-pandas-ai","name":"pandas-ai","tagline":"Chat with your database or your datalake using LLMs and RAG.","github_url":"https://github.com/sinaptik-ai/pandas-ai","owner":"sinaptik-ai","repo":"pandas-ai","owner_avatar_url":"https://avatars.githubusercontent.com/u/154438448?v=4","primary_language":"Python","stars":23746,"forks":2342,"topics":["ai","csv","data","data-analysis","data-science","data-visualization","database","datalake","gpt-4","llm","pandas","sql","text-to-sql"],"archived":false,"github_pushed_at":"2025-10-28T10:02:13+00:00","maintenance_label":"Slowing","stars_delta_30d":90,"url":"https://www.graphcanon.com/tools/sinaptik-ai-pandas-ai","markdown_url":"https://www.graphcanon.com/tools/sinaptik-ai-pandas-ai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/sinaptik-ai-pandas-ai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=sinaptik-ai-pandas-ai"}},{"type":"related","direction":"in","explanation":"DataJuicer and LangExtract both deal with processing textual data for AI applications, but DataJuicer focuses on data processing for foundation models rather than extraction.","successor_context":null,"tool":{"slug":"google-langextract","name":"langextract","tagline":"A Python library for extracting structured information from unstructured text using LLMs.","github_url":"https://github.com/google/langextract","owner":"google","repo":"langextract","owner_avatar_url":"https://avatars.githubusercontent.com/u/1342004?v=4","primary_language":"Python","stars":38400,"forks":2693,"topics":["gemini","gemini-ai","gemini-api","gemini-flash","gemini-pro","information-extration","large-language-models","llm","nlp","python","structured-data"],"archived":false,"github_pushed_at":"2026-08-11T15:31:39+00:00","maintenance_label":"Very active","stars_delta_30d":1241,"url":"https://www.graphcanon.com/tools/google-langextract","markdown_url":"https://www.graphcanon.com/tools/google-langextract.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/google-langextract","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=google-langextract"}},{"type":"integrates_with","direction":"in","explanation":"DataDreamer and DataJuicer both deal with data processing for large language models, with DataDreamer focusing on generating synthetic data for training and alignment of ML models.","successor_context":null,"tool":{"slug":"datadreamer-dev-datadreamer","name":"DataDreamer","tagline":"Prompt. Generate Synthetic Data. Train & Align Models.","github_url":"https://github.com/datadreamer-dev/DataDreamer","owner":"datadreamer-dev","repo":"DataDreamer","owner_avatar_url":"https://avatars.githubusercontent.com/u/154913957?v=4","primary_language":"Python","stars":1117,"forks":58,"topics":["alignment","deep-learning","fine-tuning","gpt","instruction-tuning","llm","llmops","llms","machine-learning","natural-language-processing","nlp","nlp-library","openai","python","pytorch","synthetic-data","synthetic-dataset-generation","transformers"],"archived":false,"github_pushed_at":"2025-02-02T21:23:50+00:00","maintenance_label":"Dormant","stars_delta_30d":2,"url":"https://www.graphcanon.com/tools/datadreamer-dev-datadreamer","markdown_url":"https://www.graphcanon.com/tools/datadreamer-dev-datadreamer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/datadreamer-dev-datadreamer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=datadreamer-dev-datadreamer"}},{"type":"related","direction":"in","explanation":"'machine-learning-for-trading' involves data processing for ML trading models, mirroring the focus of Data Juicer on improving and preparing data for foundation models.","successor_context":null,"tool":{"slug":"stefan-jansen-machine-learning-for-trading","name":"machine-learning-for-trading","tagline":"Code for Machine Learning in Trading","github_url":"https://github.com/stefan-jansen/machine-learning-for-trading","owner":"stefan-jansen","repo":"machine-learning-for-trading","owner_avatar_url":"https://avatars.githubusercontent.com/u/4275885?v=4","primary_language":"Jupyter Notebook","stars":20480,"forks":5521,"topics":["algorithmic-trading","artificial-intelligence","backtesting","data-science","deep-learning","finance","investment","investment-strategies","large-language-models","machine-learning","ml4t-workflow","polars","quantitative-finance","reinforcement-learning","synthetic-data","trading","trading-agent","trading-strategies"],"archived":false,"github_pushed_at":"2026-08-16T12:45:53+00:00","maintenance_label":"Very active","stars_delta_30d":549,"url":"https://www.graphcanon.com/tools/stefan-jansen-machine-learning-for-trading","markdown_url":"https://www.graphcanon.com/tools/stefan-jansen-machine-learning-for-trading.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/stefan-jansen-machine-learning-for-trading","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=stefan-jansen-machine-learning-for-trading"}}],"neighbours":[{"slug":"pathwaycom-llm-app","name":"llm-app","tagline":"Ready-to-run cloud templates for RAG, AI pipelines, and enterprise search with live data.","github_url":"https://github.com/pathwaycom/llm-app","owner":"pathwaycom","repo":"llm-app","owner_avatar_url":"https://avatars.githubusercontent.com/u/25750857?v=4","primary_language":"Jupyter Notebook","stars":59037,"forks":1466,"topics":["chatbot","hugging-face","llm","llm-local","llm-prompting","llm-security","llmops","machine-learning","open-ai","pathway","rag","real-time","retrieval-augmented-generation","vector-database","vector-index"],"archived":false,"github_pushed_at":"2026-07-05T17:59:07+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/pathwaycom-llm-app","markdown_url":"https://www.graphcanon.com/tools/pathwaycom-llm-app.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pathwaycom-llm-app","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pathwaycom-llm-app","shared_categories":["data-retrieval"]},{"slug":"patchy631-ai-engineering-hub","name":"ai-engineering-hub","tagline":"Tutorials on LLMs, RAGs, and real-world AI agent applications","github_url":"https://github.com/patchy631/ai-engineering-hub","owner":"patchy631","repo":"ai-engineering-hub","owner_avatar_url":"https://avatars.githubusercontent.com/u/38653995?v=4","primary_language":"Jupyter Notebook","stars":37020,"forks":6107,"topics":["agents","ai","llms","machine-learning","mcp","rag"],"archived":false,"github_pushed_at":"2026-07-27T18:43:06+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/patchy631-ai-engineering-hub","markdown_url":"https://www.graphcanon.com/tools/patchy631-ai-engineering-hub.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/patchy631-ai-engineering-hub","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=patchy631-ai-engineering-hub","shared_categories":[]},{"slug":"huggingface-datasets","name":"datasets","tagline":"Largest hub of ready-to-use datasets for AI models","github_url":"https://github.com/huggingface/datasets","owner":"huggingface","repo":"datasets","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":21791,"forks":3322,"topics":["ai","artificial-intelligence","computer-vision","dataset-hub","datasets","deep-learning","huggingface","llm","machine-learning","natural-language-processing","nlp","numpy","pandas","pytorch","speech","tensorflow"],"archived":false,"github_pushed_at":"2026-07-30T11:23:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-datasets","markdown_url":"https://www.graphcanon.com/tools/huggingface-datasets.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-datasets","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-datasets","shared_categories":["data-retrieval"]},{"slug":"unstructured-io-unstructured","name":"unstructured","tagline":"Convert documents to structured data effortlessly","github_url":"https://github.com/Unstructured-IO/unstructured","owner":"Unstructured-IO","repo":"unstructured","owner_avatar_url":"https://avatars.githubusercontent.com/u/108372208?v=4","primary_language":"HTML","stars":15238,"forks":1284,"topics":["data-pipelines","deep-learning","document-image-analysis","document-image-processing","document-parser","document-parsing","docx","donut","information-retrieval","langchain","llm","machine-learning","ml","natural-language-processing","nlp","ocr","pdf","pdf-to-json","pdf-to-text","preprocessing"],"archived":false,"github_pushed_at":"2026-07-31T20:54:17+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/unstructured-io-unstructured","markdown_url":"https://www.graphcanon.com/tools/unstructured-io-unstructured.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/unstructured-io-unstructured","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=unstructured-io-unstructured","shared_categories":["model-training","data-retrieval"]},{"slug":"data-privacy-stack-presidio","name":"presidio","tagline":"A framework for detecting and anonymizing sensitive data","github_url":"https://github.com/data-privacy-stack/presidio","owner":"data-privacy-stack","repo":"presidio","owner_avatar_url":"https://avatars.githubusercontent.com/u/275623515?v=4","primary_language":"Python","stars":10395,"forks":1237,"topics":["anonymization","data-anonymization","data-masking","data-obfuscation","data-privacy","data-redaction","de-identification","guardrails","image-redactor","named-entity-recognition","nlp","personally-identifiable-information","phi","pii","pii-detection","privacy","python","sensitive-data","spacy","transformers"],"archived":false,"github_pushed_at":"2026-08-08T21:25:09+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/data-privacy-stack-presidio","markdown_url":"https://www.graphcanon.com/tools/data-privacy-stack-presidio.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/data-privacy-stack-presidio","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=data-privacy-stack-presidio","shared_categories":["data-retrieval"]},{"slug":"mage-ai-mage-ai","name":"mage-ai","tagline":"Build, run and manage data pipelines for integrating and transforming data","github_url":"https://github.com/mage-ai/mage-ai","owner":"mage-ai","repo":"mage-ai","owner_avatar_url":"https://avatars.githubusercontent.com/u/69371472?v=4","primary_language":"Python","stars":8790,"forks":982,"topics":["artificial-intelligence","data","data-engineering","data-integration","data-pipelines","data-science","dbt","elt","etl","machine-learning","orchestration","pipeline","pipelines","python","reverse-etl","spark","sql","transformation"],"archived":false,"github_pushed_at":"2026-08-10T23:12:25+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/mage-ai-mage-ai","markdown_url":"https://www.graphcanon.com/tools/mage-ai-mage-ai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/mage-ai-mage-ai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=mage-ai-mage-ai","shared_categories":["data-retrieval"]},{"slug":"zipstack-unstract","name":"unstract","tagline":"LLM-Driven Extraction of Unstructured Data for API Deployments and ETL Pipeline Workflows","github_url":"https://github.com/Zipstack/unstract","owner":"Zipstack","repo":"unstract","owner_avatar_url":"https://avatars.githubusercontent.com/u/89070934?v=4","primary_language":"Python","stars":6932,"forks":663,"topics":["ai-agents","data-engineering","document-ai","generative-ai","idp","json-extraction","llm","mcp-server","ocr","pdf-extraction","prompt-engineering","structured-output"],"archived":false,"github_pushed_at":"2026-07-27T22:23:41+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/zipstack-unstract","markdown_url":"https://www.graphcanon.com/tools/zipstack-unstract.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zipstack-unstract","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zipstack-unstract","shared_categories":["data-retrieval"]},{"slug":"run-llama-rags","name":"rags","tagline":"Build ChatGPT over your data with natural language","github_url":"https://github.com/run-llama/rags","owner":"run-llama","repo":"rags","owner_avatar_url":"https://avatars.githubusercontent.com/u/130722866?v=4","primary_language":"Python","stars":6549,"forks":656,"topics":["agent","chatbot","chatgpt","gpts","llamaindex","llm","openai","rag","streamlit"],"archived":false,"github_pushed_at":"2024-04-05T05:36:59+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/run-llama-rags","markdown_url":"https://www.graphcanon.com/tools/run-llama-rags.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/run-llama-rags","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=run-llama-rags","shared_categories":["data-retrieval"]},{"slug":"nyldn-claude-octopus","name":"claude-octopus","tagline":"Surface AI blindspots before you ship","github_url":"https://github.com/nyldn/claude-octopus","owner":"nyldn","repo":"claude-octopus","owner_avatar_url":"https://avatars.githubusercontent.com/u/4805949?v=4","primary_language":"Shell","stars":3962,"forks":374,"topics":["ai-agents","ai-orchestration","claude-code","claude-code-plugin","codex","copilot","developer-tools","double-diamond","gemini","multi-ai","multi-llm","ollama"],"archived":false,"github_pushed_at":"2026-08-13T23:28:23+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/nyldn-claude-octopus","markdown_url":"https://www.graphcanon.com/tools/nyldn-claude-octopus.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nyldn-claude-octopus","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nyldn-claude-octopus","shared_categories":[]},{"slug":"huggingface-datatrove","name":"datatrove","tagline":"Platform-agnostic customizable pipeline processing blocks for data processing and transformation.","github_url":"https://github.com/huggingface/datatrove","owner":"huggingface","repo":"datatrove","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":3250,"forks":288,"topics":[],"archived":false,"github_pushed_at":"2026-08-06T15:27:26+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-datatrove","markdown_url":"https://www.graphcanon.com/tools/huggingface-datatrove.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-datatrove","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-datatrove","shared_categories":["model-training","data-retrieval"]},{"slug":"minimaxir-automl-gs","name":"automl-gs","tagline":"Automatically generate machine-learning models and code with input CSV and target field","github_url":"https://github.com/minimaxir/automl-gs","owner":"minimaxir","repo":"automl-gs","owner_avatar_url":"https://avatars.githubusercontent.com/u/2179708?v=4","primary_language":"Python","stars":1869,"forks":181,"topics":["automl","keras","machine-learning","python","tensorflow","xgboost"],"archived":false,"github_pushed_at":"2019-10-22T11:20:40+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/minimaxir-automl-gs","markdown_url":"https://www.graphcanon.com/tools/minimaxir-automl-gs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/minimaxir-automl-gs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=minimaxir-automl-gs","shared_categories":["model-training","data-retrieval"]},{"slug":"huggingface-aisheets","name":"aisheets","tagline":"Build, enrich, and transform datasets using AI models with no code","github_url":"https://github.com/huggingface/aisheets","owner":"huggingface","repo":"aisheets","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"TypeScript","stars":1638,"forks":140,"topics":["ai","llm-evaluation","llms","nocode","oss","synthetic-data"],"archived":false,"github_pushed_at":"2026-05-26T10:33:23+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/huggingface-aisheets","markdown_url":"https://www.graphcanon.com/tools/huggingface-aisheets.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-aisheets","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-aisheets","shared_categories":["data-retrieval"]}]}}