{"data":{"slug":"dineshsoudagar-local-llms-on-android","name":"local-llms-on-android","tagline":"Run local LLMs for offline chat and question answering on Android.","github_url":"https://github.com/dineshsoudagar/local-llms-on-android","owner":"dineshsoudagar","repo":"local-llms-on-android","owner_avatar_url":"https://avatars.githubusercontent.com/u/75389628?v=4","primary_language":"Kotlin","stars":419,"forks":53,"topics":["android","android-app","chatbot","gemma4","gemma4-2b","gemma4-e4b","huggingface-tokenizers","litert","litert-lm","llama3","local-llm","local-llm-integration","mobile-ai","offline-inference","on-device-ai","onnx-runtime","qwen"],"archived":false,"github_pushed_at":"2026-08-13T05:49:16+00:00","maintenance_label":"Steady","stars_delta_30d":38,"url":"https://www.graphcanon.com/tools/dineshsoudagar-local-llms-on-android","markdown_url":"https://www.graphcanon.com/tools/dineshsoudagar-local-llms-on-android.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/dineshsoudagar-local-llms-on-android","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=dineshsoudagar-local-llms-on-android","description":"Run local LLMs like Gemma, Qwen, and LLaMA on Android for offline, private, real-time chat and question answering with LiteRT and ONNX Runtime.","homepage_url":null,"license":"MIT","open_issues":12,"watchers":8,"ai_summary":"Enables execution of Large Language Models such as Gemma, Qwen, and LLaMA on Android devices via LiteRT and ONNX Runtime for private use cases.","readme_excerpt":"## ⚙️ Requirements\n\n- [Android Studio](https://developer.android.com/studio)\n- A physical Android device for deployment and testing\n- 4 GB or more RAM for smaller models\n- More RAM is recommended for larger models such as **Gemma 4 E2B** and **Gemma 4 E4B**\n- A temporary internet connection for downloading models inside the app\n- Real hardware is preferred; emulators are mainly useful for UI checks\n\n---","github_created_at":"2025-05-04T15:04:09+00:00","created_at":"2026-07-15T10:59:40.868924+00:00","updated_at":"2026-09-20T05:03:11.564837+00:00","categories":[{"slug":"inference-serving","name":"Inference & Serving","url":"https://www.graphcanon.com/categories/inference-serving","markdown_url":"https://www.graphcanon.com/categories/inference-serving.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/inference-serving"},{"slug":"model-training","name":"Model Training","url":"https://www.graphcanon.com/categories/model-training","markdown_url":"https://www.graphcanon.com/categories/model-training.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/model-training"}],"tags":[{"slug":"android","name":"android"},{"slug":"chatbot","name":"chatbot"},{"slug":"gemma4","name":"gemma4"},{"slug":"huggingface-tokenizers","name":"huggingface-tokenizers"},{"slug":"litert","name":"litert"},{"slug":"llama3","name":"llama3"},{"slug":"onnx-runtime","name":"onnx-runtime"},{"slug":"qwen","name":"qwen"}],"trust":{"provenance":{"is_fork":false,"github_id":977593093,"owner_type":"User","methodology":"github_public_v1","parent_repo":null,"near_duplicate_slugs":[]},"computed_at":"2026-09-20T05:03:09.732Z","maintenance":{"label":"Steady","score":60,"methodology":"github_public_v1","releases_90d":0,"days_since_push":37,"last_release_at":"2026-04-26T04:48:52Z","stars_delta_30d":38,"open_issues_delta_30d":0},"security_summary":{"status":"no_lockfile","scanner":null,"low_count":0,"high_count":0,"last_scan_at":"2026-07-15T10:59:42.064Z","medium_count":0,"scan_profile":"none","critical_count":0}},"capability_facts":{"scan":{"source":"repo_scan","observed_at":"2026-09-20T05:03:10.767Z"},"languages":{"value":["kotlin"],"source":"github.language","observed_at":"2026-09-20T05:03:10.767Z"},"license_spdx":{"value":"MIT","source":"github.license","observed_at":"2026-09-20T05:03:10.767Z"}},"decision_facts":{"hosting":null,"pricing":null,"requirements":{"notes":["Supports deployment on Android devices using Kotlin programming language with LiteRT engine and ONNX Runtime for inference."]},"constraints":null,"when_to_use":["To enable offline chat and question answering capabilities, particularly when Gemma, Qwen, or LLaMA are the preferred model architectures.","When there is a need to ensure data privacy by processing user inputs locally without internet connectivity."],"when_not_to_use":["If real-time connection with internet-based services is necessary for chatbot operation and interaction.","In scenarios where the model's computational requirements exceed the capabilities of the target Android device, possibly leading to performance issues or excessive battery drain."],"source":"enrich:decision_facts","observed_at":"2026-07-17T06:07:19.180Z"},"constraint_facets":null,"decision_summary":[{"label":"Requirements","value":"Supports deployment on Android devices using Kotlin programming language with LiteRT engine and ONNX Runtime for inference."},{"label":"Adopt for","value":"local-llms-on-android enables offline deployment and execution of large language models on Android devices using LiteRT and ONNX Runtime for applications requiring real-time conversation."}]}}