{"data":{"node":{"slug":"fivetran-great-expectations","name":"great_expectations","tagline":"Always know what to expect from your data","github_url":"https://github.com/fivetran/great_expectations","owner":"fivetran","repo":"great_expectations","owner_avatar_url":"https://avatars.githubusercontent.com/u/2722259?v=4","primary_language":"Python","stars":11690,"forks":1790,"topics":["cleandata","data-engineering","data-profilers","data-profiling","data-quality","data-science","data-unit-tests","datacleaner","datacleaning","dataquality","dataunittest","eda","exploratory-analysis","exploratory-data-analysis","exploratorydataanalysis","mlops","pipeline","pipeline-debt","pipeline-testing","pipeline-tests"],"archived":false,"github_pushed_at":"2026-08-02T02:39:28+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/fivetran-great-expectations","markdown_url":"https://www.graphcanon.com/tools/fivetran-great-expectations.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fivetran-great-expectations","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fivetran-great-expectations"},"categories":[{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"}],"tags":[{"slug":"data-engineering","name":"data-engineering"},{"slug":"data-quality","name":"data-quality"},{"slug":"exploratory-data-analysis","name":"exploratory-data-analysis"},{"slug":"mlops","name":"mlops"}],"edges":[],"neighbours":[{"slug":"sinaptik-ai-pandas-ai","name":"pandas-ai","tagline":"Chat with your database or your datalake using LLMs and RAG.","github_url":"https://github.com/sinaptik-ai/pandas-ai","owner":"sinaptik-ai","repo":"pandas-ai","owner_avatar_url":"https://avatars.githubusercontent.com/u/154438448?v=4","primary_language":"Python","stars":23746,"forks":2342,"topics":["ai","csv","data","data-analysis","data-science","data-visualization","database","datalake","gpt-4","llm","pandas","sql","text-to-sql"],"archived":false,"github_pushed_at":"2025-10-28T10:02:13+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/sinaptik-ai-pandas-ai","markdown_url":"https://www.graphcanon.com/tools/sinaptik-ai-pandas-ai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/sinaptik-ai-pandas-ai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=sinaptik-ai-pandas-ai","shared_categories":["data-retrieval"]},{"slug":"huggingface-datasets","name":"datasets","tagline":"Largest hub of ready-to-use datasets for AI models","github_url":"https://github.com/huggingface/datasets","owner":"huggingface","repo":"datasets","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":21791,"forks":3322,"topics":["ai","artificial-intelligence","computer-vision","dataset-hub","datasets","deep-learning","huggingface","llm","machine-learning","natural-language-processing","nlp","numpy","pandas","pytorch","speech","tensorflow"],"archived":false,"github_pushed_at":"2026-07-30T11:23:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-datasets","markdown_url":"https://www.graphcanon.com/tools/huggingface-datasets.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-datasets","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-datasets","shared_categories":["data-retrieval"]},{"slug":"canner-wrenai","name":"WrenAI","tagline":"GenBI for AI agents, turns natural-language questions into trusted dashboards and SQL","github_url":"https://github.com/Canner/WrenAI","owner":"Canner","repo":"WrenAI","owner_avatar_url":"https://avatars.githubusercontent.com/u/7250217?v=4","primary_language":"Python","stars":17295,"forks":1955,"topics":["ai-agents","bigquery","business-intelligence","charts","clickhouse","context-engineering","dashboard","databricks","duckdb","genbi","generative-ai","llm","mcp","postgresql","rag","semantic-layer","snowflake","sql","text-to-sql","text2sql"],"archived":false,"github_pushed_at":"2026-08-18T05:19:19+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/canner-wrenai","markdown_url":"https://www.graphcanon.com/tools/canner-wrenai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/canner-wrenai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=canner-wrenai","shared_categories":["data-retrieval"]},{"slug":"mage-ai-mage-ai","name":"mage-ai","tagline":"Build, run and manage data pipelines for integrating and transforming data","github_url":"https://github.com/mage-ai/mage-ai","owner":"mage-ai","repo":"mage-ai","owner_avatar_url":"https://avatars.githubusercontent.com/u/69371472?v=4","primary_language":"Python","stars":8790,"forks":982,"topics":["artificial-intelligence","data","data-engineering","data-integration","data-pipelines","data-science","dbt","elt","etl","machine-learning","orchestration","pipeline","pipelines","python","reverse-etl","spark","sql","transformation"],"archived":false,"github_pushed_at":"2026-08-10T23:12:25+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/mage-ai-mage-ai","markdown_url":"https://www.graphcanon.com/tools/mage-ai-mage-ai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/mage-ai-mage-ai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=mage-ai-mage-ai","shared_categories":["data-retrieval"]},{"slug":"evidentlyai-evidently","name":"evidently","tagline":"An open-source ML and LLM observability framework.","github_url":"https://github.com/evidentlyai/evidently","owner":"evidentlyai","repo":"evidently","owner_avatar_url":"https://avatars.githubusercontent.com/u/75031056?v=4","primary_language":"Jupyter Notebook","stars":7790,"forks":895,"topics":["data-drift","data-quality","data-science","data-validation","generative-ai","hacktoberfest","html-report","jupyter-notebook","llm","llmops","machine-learning","mlops","model-monitoring","pandas-dataframe"],"archived":false,"github_pushed_at":"2026-08-05T16:29:57+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/evidentlyai-evidently","markdown_url":"https://www.graphcanon.com/tools/evidentlyai-evidently.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/evidentlyai-evidently","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=evidentlyai-evidently","shared_categories":[]},{"slug":"alteryx-featuretools","name":"featuretools","tagline":"An open source python library for automated feature engineering","github_url":"https://github.com/alteryx/featuretools","owner":"alteryx","repo":"featuretools","owner_avatar_url":"https://avatars.githubusercontent.com/u/12972388?v=4","primary_language":"Python","stars":7665,"forks":915,"topics":["automated-feature-engineering","automated-machine-learning","automl","data-science","feature-engineering","machine-learning","python","scikit-learn"],"archived":false,"github_pushed_at":"2026-07-27T13:04:38+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/alteryx-featuretools","markdown_url":"https://www.graphcanon.com/tools/alteryx-featuretools.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alteryx-featuretools","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alteryx-featuretools","shared_categories":[]},{"slug":"datajuicer-data-juicer","name":"data-juicer","tagline":"Data processing for and with foundation models","github_url":"https://github.com/datajuicer/data-juicer","owner":"datajuicer","repo":"data-juicer","owner_avatar_url":"https://avatars.githubusercontent.com/u/223222708?v=4","primary_language":"Python","stars":6897,"forks":404,"topics":["data","data-analysis","data-pipeline","data-processing","data-science","data-visualization","foundation-models","instruction-tuning","large-language-models","llm","llms","multi-modal","pre-training","synthetic-data"],"archived":false,"github_pushed_at":"2026-08-13T09:19:31+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/datajuicer-data-juicer","markdown_url":"https://www.graphcanon.com/tools/datajuicer-data-juicer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/datajuicer-data-juicer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=datajuicer-data-juicer","shared_categories":["data-retrieval"]},{"slug":"ucbepic-docetl","name":"docetl","tagline":"A system for agentic LLM-powered data processing and ETL","github_url":"https://github.com/ucbepic/docetl","owner":"ucbepic","repo":"docetl","owner_avatar_url":"https://avatars.githubusercontent.com/u/88680502?v=4","primary_language":"Python","stars":3961,"forks":421,"topics":["agents","data","data-pipelines","document-analysis","document-processing","elt","etl","llm","python","semantic-data","unstructured-data","unstructured-data-analysis","workflow"],"archived":false,"github_pushed_at":"2026-08-09T23:31:04+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/ucbepic-docetl","markdown_url":"https://www.graphcanon.com/tools/ucbepic-docetl.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ucbepic-docetl","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ucbepic-docetl","shared_categories":["data-retrieval"]},{"slug":"ploomber-ploomber","name":"ploomber","tagline":"The fastest way to build data pipelines. Develop iteratively, deploy anywhere.","github_url":"https://github.com/ploomber/ploomber","owner":"ploomber","repo":"ploomber","owner_avatar_url":"https://avatars.githubusercontent.com/u/60114551?v=4","primary_language":"Python","stars":3622,"forks":243,"topics":["data-engineering","data-science","jupyter","jupyter-notebooks","machine-learning","mlops","notebooks","papermill","pipelines","pycharm","vscode","workflow"],"archived":true,"github_pushed_at":"2025-05-29T22:02:03+00:00","maintenance_label":"Archived","url":"https://www.graphcanon.com/tools/ploomber-ploomber","markdown_url":"https://www.graphcanon.com/tools/ploomber-ploomber.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ploomber-ploomber","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ploomber-ploomber","shared_categories":[]},{"slug":"huggingface-datatrove","name":"datatrove","tagline":"Platform-agnostic customizable pipeline processing blocks for data processing and transformation.","github_url":"https://github.com/huggingface/datatrove","owner":"huggingface","repo":"datatrove","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":3250,"forks":288,"topics":[],"archived":false,"github_pushed_at":"2026-08-06T15:27:26+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-datatrove","markdown_url":"https://www.graphcanon.com/tools/huggingface-datatrove.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-datatrove","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-datatrove","shared_categories":["data-retrieval"]},{"slug":"spiceai-spiceai","name":"spiceai","tagline":"A real-time analytics node for data-grounded AI applications","github_url":"https://github.com/spiceai/spiceai","owner":"spiceai","repo":"spiceai","owner_avatar_url":"https://avatars.githubusercontent.com/u/73862742?v=4","primary_language":"Rust","stars":3069,"forks":222,"topics":["artificial-intelligence","data","data-federation","developers","full-text-search","infrastructure","llm-inference","machine-learning","sql"],"archived":false,"github_pushed_at":"2026-08-24T17:16:22+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/spiceai-spiceai","markdown_url":"https://www.graphcanon.com/tools/spiceai-spiceai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/spiceai-spiceai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=spiceai-spiceai","shared_categories":["data-retrieval"]},{"slug":"minimaxir-automl-gs","name":"automl-gs","tagline":"Automatically generate machine-learning models and code with input CSV and target field","github_url":"https://github.com/minimaxir/automl-gs","owner":"minimaxir","repo":"automl-gs","owner_avatar_url":"https://avatars.githubusercontent.com/u/2179708?v=4","primary_language":"Python","stars":1869,"forks":181,"topics":["automl","keras","machine-learning","python","tensorflow","xgboost"],"archived":false,"github_pushed_at":"2019-10-22T11:20:40+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/minimaxir-automl-gs","markdown_url":"https://www.graphcanon.com/tools/minimaxir-automl-gs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/minimaxir-automl-gs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=minimaxir-automl-gs","shared_categories":["data-retrieval"]}]}}