{"data":{"slug":"nvidia-generativeaiexamples","name":"GenerativeAIExamples","tagline":"Generative AI reference workflows for accelerated infrastructure and microservice architecture","github_url":"https://github.com/NVIDIA/GenerativeAIExamples","owner":"NVIDIA","repo":"GenerativeAIExamples","owner_avatar_url":"https://avatars.githubusercontent.com/u/1728152?v=4","primary_language":"Jupyter Notebook","stars":4149,"forks":1095,"topics":["gpu-acceleration","large-language-models","llm","llm-inference","microservice","nemo","rag","retrieval-augmented-generation","tensorrt","triton-inference-server"],"archived":false,"github_pushed_at":"2026-08-05T16:54:49+00:00","maintenance_label":"Active","stars_delta_30d":29,"url":"https://www.graphcanon.com/tools/nvidia-generativeaiexamples","markdown_url":"https://www.graphcanon.com/tools/nvidia-generativeaiexamples.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-generativeaiexamples","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-generativeaiexamples","description":"Generative AI reference workflows optimized for accelerated infrastructure and microservice architecture.","homepage_url":null,"license":"Apache-2.0","open_issues":86,"watchers":89,"ai_summary":"A collection of Jupyter Notebook-based examples that showcase how to deploy and operate generative AI models with GPU-acceleration and microservices on platforms like NVIDIA's TensorRT, Triton Inference Server, and LangChain.","readme_excerpt":"### RAG with Local NIM Deployment and LangChain\n\n- Tips for Building a RAG Pipeline with NVIDIA AI LangChain AI Endpoints by Amit Bleiweiss. [[Blog](https://developer.nvidia.com/blog/tips-for-building-a-rag-pipeline-with-nvidia-ai-langchain-ai-endpoints/), [Notebook](https://github.com/NVIDIA/GenerativeAIExamples/blob/v0.7.0/notebooks/08_RAG_Langchain_with_Local_NIM.ipynb)]\n\nFor more information, refer to the [Generative AI Example releases](https://github.com/NVIDIA/GenerativeAIExamples/releases/).\n\n---\n\n### Getting Started\n\n- [Prerequisites](./docs/common-prerequisites.md)","github_created_at":"2023-10-19T13:46:31+00:00","created_at":"2026-07-07T17:35:29.5725+00:00","updated_at":"2026-08-17T18:01:25.686738+00:00","categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"},{"slug":"llm-frameworks","name":"LLM Frameworks","url":"https://www.graphcanon.com/categories/llm-frameworks","markdown_url":"https://www.graphcanon.com/categories/llm-frameworks.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/llm-frameworks"}],"tags":[{"slug":"gpu-acceleration","name":"gpu acceleration"},{"slug":"large-language-models","name":"large language models"},{"slug":"llm-inference","name":"llm-inference"},{"slug":"microservice","name":"microservice"},{"slug":"nemo","name":"nemo"},{"slug":"rag","name":"rag"},{"slug":"retrieval-augmented-generation","name":"retrieval-augmented-generation"},{"slug":"tensorrt","name":"tensorrt"}],"trust":{"provenance":{"is_fork":false,"github_id":707237272,"owner_type":"Organization","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-08-17T18:01:23.219Z","maintenance":{"label":"Active","score":82,"methodology":"github_public_v1","releases_90d":0,"days_since_push":12,"last_release_at":"2024-08-21T03:11:58Z","stars_delta_30d":29,"open_issues_delta_30d":1},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-11T11:05:32.113Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-08-17T18:01:24.438Z"},"languages":{"value":["jupyter notebook"],"source":"github.language","observed_at":"2026-08-17T18:01:24.438Z"},"license_spdx":{"value":"Apache-2.0","source":"github.license","observed_at":"2026-08-17T18:01:24.438Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":null,"constraints":null,"when_to_use":["To accelerate deployment of generative AI on GPU-supported infrastructure","For leveraging NVIDIA's TensorRT or Triton Inference Server specifically","When building RAG pipelines with LangChain endpoints"],"when_not_to_use":["If preferred platform is not aligned with NVIDIA's offerings","In cases where deployment outside microservice architecture is needed","For scenarios that do not require GPU acceleration or Triton Inference Server integration"],"source":"enrich:decision_facts","observed_at":"2026-07-14T20:56:50.349Z"},"constraint_facets":null,"decision_summary":[{"label":"Adopt for","value":"Jupyter Notebook-based reference workflows for GPU-accelerated and microservice-oriented deployment of generative AI models, using platforms like NVIDIA TensorRT and Triton Inference Server."}]}}