{"data":{"slug":"neumtry-neumai","name":"NeumAI","tagline":"Framework to manage creation and synchronization of vector embeddings at large scale","github_url":"https://github.com/NeumTry/NeumAI","owner":"NeumTry","repo":"NeumAI","owner_avatar_url":"https://avatars.githubusercontent.com/u/129831068?v=4","primary_language":"Python","stars":867,"forks":50,"topics":["ai","chatgpt","data","data-engineering","database","embeddings","etl","llm","llmops","mlops","ops","pipeline","python","rag","retrieval","vector-database","vectors"],"archived":false,"github_pushed_at":"2024-01-15T23:00:58+00:00","maintenance_label":"Dormant","stars_delta_30d":3,"url":"https://www.graphcanon.com/tools/neumtry-neumai","markdown_url":"https://www.graphcanon.com/tools/neumtry-neumai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/neumtry-neumai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=neumtry-neumai","description":"Neum AI is a best-in-class framework to manage the creation and synchronization of vector embeddings at large scale.","homepage_url":"https://neum.ai","license":"Apache-2.0","open_issues":9,"watchers":6,"ai_summary":"Neum AI provides tools for managing the generation, storage, and synchronization of vector embeddings in data engineering environments. It includes support for pipeline operations, embedding retrieval, and integration with various machine learning workflows.","readme_excerpt":"### Self-host\n\nIf you are interested in deploying Neum AI to your own cloud contact us at [founders@tryneum.com](mailto:founders@tryneum.com).\n\nWe have a sample backend architecture published on [GitHub](https://github.com/NeumTry/neum-at-scale) which you can use as a starting point.","github_created_at":"2023-09-14T00:04:50+00:00","created_at":"2026-07-07T17:43:12.802106+00:00","updated_at":"2026-08-21T00:01:10.637959+00:00","categories":[{"slug":"data-retrieval","name":"Data & Retrieval","url":"https://www.graphcanon.com/categories/data-retrieval","markdown_url":"https://www.graphcanon.com/categories/data-retrieval.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/data-retrieval"},{"slug":"vector-databases","name":"Vector Databases","url":"https://www.graphcanon.com/categories/vector-databases","markdown_url":"https://www.graphcanon.com/categories/vector-databases.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/vector-databases"}],"tags":[{"slug":"ai","name":"ai"},{"slug":"data-engineering","name":"data-engineering"},{"slug":"database","name":"database"},{"slug":"embeddings","name":"embeddings"},{"slug":"etl","name":"etl"},{"slug":"llmops","name":"llmops"},{"slug":"mlops","name":"mlops"},{"slug":"pipeline","name":"pipeline"}],"trust":{"provenance":{"is_fork":false,"github_id":691317919,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-08-21T00:01:09.753Z","maintenance":{"label":"Dormant","score":18,"methodology":"github_public_v1","releases_90d":0,"days_since_push":948,"last_release_at":"2024-01-07T17:38:43Z","stars_delta_30d":3,"open_issues_delta_30d":0},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-11T11:23:08.598Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-08-21T00:01:10.248Z"},"languages":{"value":["python"],"source":"github.language","observed_at":"2026-08-21T00:01:10.248Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-08-21T00:01:10.248Z"}},"decision_facts":{"hosting":null,"pricing":{"model":"freemium","summary":"Offers an open-source model under the Apache-2.0 license, potentially featuring a free-tier with premium/support options."},"requirements":{"notes":["Requires Python and compatibility with infrastructure that supports its backend architecture.","Contact their team at founders@tryneum.com for self-hosting."],"min_ram_gb":null,"requires_docker":false},"constraints":{"min_ram_gb":null,"pricing_model":"freemium","requires_docker":false},"when_to_use":["When you require robust and scalable infrastructure specifically designed for creating and synchronizing vector embeddings at scale.","If your project involves implementing a retrieval-augmented generation pipeline, where efficient embedding management is critical.","For users interested in having full control over their deployment environment, as NeumAI supports self-hosting and offers sample backend architectures to get you started."],"when_not_to_use":["When your project requires customization beyond what the provided architecture allows, without the support expected from commercial offerings or competitive open-source frameworks.","If your needs are simpler and don't demand large-scale operations, NeumAI’s capabilities focused on handling vast vector sets may be excessive for smaller projects."],"source":"enrich:decision_facts","observed_at":"2026-07-11T03:41:00.422Z"},"constraint_facets":{"min_ram_gb":null,"pricing_model":"freemium","requires_docker":false},"decision_summary":[{"label":"Pricing","value":"freemium - Offers an open-source model under the Apache-2.0 license, potentially featuring a free-tier with premium/support options."},{"label":"Requirements","value":"Requires Python and compatibility with infrastructure that supports its backend architecture.; Contact their team at founders@tryneum.com for self-hosting."},{"label":"Adopt for","value":"NeumAI stands out in the space of managing large-scale vector embeddings, offering tools tailored for operations such as retrieval-augmented generation (RAG). Users looking to self-host embeddings management with an open"}]}}