{"data":{"node":{"slug":"upgini-upgini","name":"upgini","tagline":"Data search & enrichment library for Machine Learning","github_url":"https://github.com/upgini/upgini","owner":"upgini","repo":"upgini","owner_avatar_url":"https://avatars.githubusercontent.com/u/48863731?v=4","primary_language":"Python","stars":355,"forks":26,"topics":["automated-feature-engineering","automl","automl-pipeline","chatgpt","data-enrichment","data-science","feature-engineering","feature-extraction","feature-selection","features","kaggle","kaggle-solution","large-language-models","llm","machine-learning","open-data","open-datasets","public-data","python-library","scikit-learn"],"archived":false,"github_pushed_at":"2026-07-30T11:34:09+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/upgini-upgini","markdown_url":"https://www.graphcanon.com/tools/upgini-upgini.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/upgini-upgini","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=upgini-upgini"},"categories":[{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"},{"slug":"model-training","name":"Model Training","url":"https://www.graphcanon.com/categories/model-training","markdown_url":"https://www.graphcanon.com/categories/model-training.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/model-training"}],"tags":[{"slug":"automated-feature-engineering","name":"automated-feature-engineering"},{"slug":"automl","name":"automl"},{"slug":"chatgpt","name":"chatgpt"},{"slug":"data-enrichment","name":"data-enrichment"},{"slug":"feature-extraction","name":"feature-extraction"},{"slug":"kaggle","name":"kaggle"},{"slug":"large-language-models","name":"large language models"},{"slug":"llm","name":"llm"}],"edges":[],"neighbours":[{"slug":"pathwaycom-llm-app","name":"llm-app","tagline":"Ready-to-run cloud templates for RAG, AI pipelines, and enterprise search with live data.","github_url":"https://github.com/pathwaycom/llm-app","owner":"pathwaycom","repo":"llm-app","owner_avatar_url":"https://avatars.githubusercontent.com/u/25750857?v=4","primary_language":"Jupyter Notebook","stars":59037,"forks":1466,"topics":["chatbot","hugging-face","llm","llm-local","llm-prompting","llm-security","llmops","machine-learning","open-ai","pathway","rag","real-time","retrieval-augmented-generation","vector-database","vector-index"],"archived":false,"github_pushed_at":"2026-07-05T17:59:07+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/pathwaycom-llm-app","markdown_url":"https://www.graphcanon.com/tools/pathwaycom-llm-app.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pathwaycom-llm-app","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pathwaycom-llm-app","shared_categories":["data-retrieval"]},{"slug":"huggingface-datasets","name":"datasets","tagline":"Largest hub of ready-to-use datasets for AI models","github_url":"https://github.com/huggingface/datasets","owner":"huggingface","repo":"datasets","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":21791,"forks":3322,"topics":["ai","artificial-intelligence","computer-vision","dataset-hub","datasets","deep-learning","huggingface","llm","machine-learning","natural-language-processing","nlp","numpy","pandas","pytorch","speech","tensorflow"],"archived":false,"github_pushed_at":"2026-07-30T11:23:49+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-datasets","markdown_url":"https://www.graphcanon.com/tools/huggingface-datasets.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-datasets","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-datasets","shared_categories":["data-retrieval"]},{"slug":"canner-wrenai","name":"WrenAI","tagline":"GenBI for AI agents, turns natural-language questions into trusted dashboards and SQL","github_url":"https://github.com/Canner/WrenAI","owner":"Canner","repo":"WrenAI","owner_avatar_url":"https://avatars.githubusercontent.com/u/7250217?v=4","primary_language":"Python","stars":17295,"forks":1955,"topics":["ai-agents","bigquery","business-intelligence","charts","clickhouse","context-engineering","dashboard","databricks","duckdb","genbi","generative-ai","llm","mcp","postgresql","rag","semantic-layer","snowflake","sql","text-to-sql","text2sql"],"archived":false,"github_pushed_at":"2026-08-18T05:19:19+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/canner-wrenai","markdown_url":"https://www.graphcanon.com/tools/canner-wrenai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/canner-wrenai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=canner-wrenai","shared_categories":["data-retrieval"]},{"slug":"zilliztech-deep-searcher","name":"deep-searcher","tagline":"Open Source Deep Research Alternative to Reason and Search on Private Data.","github_url":"https://github.com/zilliztech/deep-searcher","owner":"zilliztech","repo":"deep-searcher","owner_avatar_url":"https://avatars.githubusercontent.com/u/18416694?v=4","primary_language":"Python","stars":8060,"forks":775,"topics":["agent","agentic-rag","claude","deep-research","deepseek","deepseek-r1","grok","grok3","llama4","llm","milvus","openai","qwen3","rag","reasoning-models","vector-database","zilliz"],"archived":false,"github_pushed_at":"2025-11-19T06:04:16+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/zilliztech-deep-searcher","markdown_url":"https://www.graphcanon.com/tools/zilliztech-deep-searcher.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zilliztech-deep-searcher","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zilliztech-deep-searcher","shared_categories":[]},{"slug":"datajuicer-data-juicer","name":"data-juicer","tagline":"Data processing for and with foundation models","github_url":"https://github.com/datajuicer/data-juicer","owner":"datajuicer","repo":"data-juicer","owner_avatar_url":"https://avatars.githubusercontent.com/u/223222708?v=4","primary_language":"Python","stars":6897,"forks":404,"topics":["data","data-analysis","data-pipeline","data-processing","data-science","data-visualization","foundation-models","instruction-tuning","large-language-models","llm","llms","multi-modal","pre-training","synthetic-data"],"archived":false,"github_pushed_at":"2026-08-13T09:19:31+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/datajuicer-data-juicer","markdown_url":"https://www.graphcanon.com/tools/datajuicer-data-juicer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/datajuicer-data-juicer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=datajuicer-data-juicer","shared_categories":["model-training","data-retrieval"]},{"slug":"huggingface-datatrove","name":"datatrove","tagline":"Platform-agnostic customizable pipeline processing blocks for data processing and transformation.","github_url":"https://github.com/huggingface/datatrove","owner":"huggingface","repo":"datatrove","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":3250,"forks":288,"topics":[],"archived":false,"github_pushed_at":"2026-08-06T15:27:26+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-datatrove","markdown_url":"https://www.graphcanon.com/tools/huggingface-datatrove.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-datatrove","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-datatrove","shared_categories":["model-training","data-retrieval"]},{"slug":"spiceai-spiceai","name":"spiceai","tagline":"A real-time analytics node for data-grounded AI applications","github_url":"https://github.com/spiceai/spiceai","owner":"spiceai","repo":"spiceai","owner_avatar_url":"https://avatars.githubusercontent.com/u/73862742?v=4","primary_language":"Rust","stars":3069,"forks":222,"topics":["artificial-intelligence","data","data-federation","developers","full-text-search","infrastructure","llm-inference","machine-learning","sql"],"archived":false,"github_pushed_at":"2026-08-24T17:16:22+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/spiceai-spiceai","markdown_url":"https://www.graphcanon.com/tools/spiceai-spiceai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/spiceai-spiceai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=spiceai-spiceai","shared_categories":["data-retrieval"]},{"slug":"devflowinc-trieve","name":"trieve","tagline":"All-in-one platform for search recommendations RAG analytics offered via API","github_url":"https://github.com/devflowinc/trieve","owner":"devflowinc","repo":"trieve","owner_avatar_url":"https://avatars.githubusercontent.com/u/122049913?v=4","primary_language":"Rust","stars":2716,"forks":251,"topics":["actix","actix-web","ai","artificial-intelligence","diesel","embedding","hacktoberfest","llm","postgresql","qdrant","qdrant-vector-database","rag","retrieval-augmented-generation","rust","search","search-engine","solidjs","tailwindcss","vector-search"],"archived":false,"github_pushed_at":"2026-01-25T23:25:46+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/devflowinc-trieve","markdown_url":"https://www.graphcanon.com/tools/devflowinc-trieve.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/devflowinc-trieve","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=devflowinc-trieve","shared_categories":["data-retrieval"]},{"slug":"agentset-ai-agentset","name":"agentset","tagline":"The open-source RAG platform with built-in citations and support for deep research","github_url":"https://github.com/agentset-ai/agentset","owner":"agentset-ai","repo":"agentset","owner_avatar_url":"https://avatars.githubusercontent.com/u/200139246?v=4","primary_language":"TypeScript","stars":2066,"forks":185,"topics":["agentic-rag","ai","ai-agents","ai-sdk","chatbots","embeddings","genai","llms","memory","memory-management","rag","vercel-ai-sdk"],"archived":false,"github_pushed_at":"2026-07-16T13:11:34+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/agentset-ai-agentset","markdown_url":"https://www.graphcanon.com/tools/agentset-ai-agentset.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/agentset-ai-agentset","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=agentset-ai-agentset","shared_categories":["data-retrieval"]},{"slug":"huggingface-aisheets","name":"aisheets","tagline":"Build, enrich, and transform datasets using AI models with no code","github_url":"https://github.com/huggingface/aisheets","owner":"huggingface","repo":"aisheets","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"TypeScript","stars":1638,"forks":140,"topics":["ai","llm-evaluation","llms","nocode","oss","synthetic-data"],"archived":false,"github_pushed_at":"2026-05-26T10:33:23+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/huggingface-aisheets","markdown_url":"https://www.graphcanon.com/tools/huggingface-aisheets.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-aisheets","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-aisheets","shared_categories":["data-retrieval"]},{"slug":"akshata29-entaoai","name":"entaoai","tagline":"Accelerator for uploading enterprise data and using OpenAI services to interact with it.","github_url":"https://github.com/akshata29/entaoai","owner":"akshata29","repo":"entaoai","owner_avatar_url":"https://avatars.githubusercontent.com/u/18509807?v=4","primary_language":"TypeScript","stars":866,"forks":245,"topics":["azure","azure-functions","azure-openai","azure-webapp","azureopenai","chatgpt","cognitive-search","gpt-3","gpt-35-turbo","langchain","openai","pinecone","redis-search","vector-store"],"archived":false,"github_pushed_at":"2025-01-02T16:23:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/akshata29-entaoai","markdown_url":"https://www.graphcanon.com/tools/akshata29-entaoai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/akshata29-entaoai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=akshata29-entaoai","shared_categories":["data-retrieval"]},{"slug":"celadaniel-free-ai-resources-x","name":"free-ai-resources-x","tagline":"A curated collection of free AI resources","github_url":"https://github.com/CelaDaniel/free-ai-resources-x","owner":"CelaDaniel","repo":"free-ai-resources-x","owner_avatar_url":"https://avatars.githubusercontent.com/u/13622306?v=4","primary_language":null,"stars":709,"forks":102,"topics":["ai","ai-agents","ai-courses","ai-ethics","ai-tools","artificial-intelligence","awesome","awesome-list","computer-vision","data-science","deep-learning","free-resources","generative-ai","large-language-models","learning-resources","llm","machine-learning","nlp","prompt-engineering","reinforcement-learning"],"archived":false,"github_pushed_at":"2026-05-21T19:29:45+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/celadaniel-free-ai-resources-x","markdown_url":"https://www.graphcanon.com/tools/celadaniel-free-ai-resources-x.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/celadaniel-free-ai-resources-x","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=celadaniel-free-ai-resources-x","shared_categories":["model-training"]}]}}