{"data":{"node":{"slug":"index-tts-index-tts","name":"index-tts","tagline":"A Breakthrough in Emotionally Expressive and Duration-Controlled Auto-Regressive Zero-Shot Text-to-Speech","github_url":"https://github.com/index-tts/index-tts","owner":"index-tts","repo":"index-tts","owner_avatar_url":"https://avatars.githubusercontent.com/u/196291161?v=4","primary_language":"Python","stars":22239,"forks":2709,"topics":["bigvgan","cross-lingual","indextts","text-to-speech","tts","voice-clone","zero-shot-tts"],"archived":false,"github_pushed_at":"2026-07-14T11:43:37+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/index-tts-index-tts","markdown_url":"https://www.graphcanon.com/tools/index-tts-index-tts.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/index-tts-index-tts","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=index-tts-index-tts"},"categories":[{"slug":"speech-audio","name":"Speech & Audio","url":"https://www.graphcanon.com/categories/speech-audio","markdown_url":"https://www.graphcanon.com/categories/speech-audio.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/speech-audio"}],"tags":[{"slug":"bigvgan","name":"bigvgan"},{"slug":"cross-lingual","name":"cross-lingual"},{"slug":"indextts","name":"indextts"},{"slug":"text-to-speech","name":"text-to-speech"},{"slug":"tts","name":"tts"},{"slug":"voice-clone","name":"voice-clone"},{"slug":"zero-shot-tts","name":"zero-shot-tts"}],"edges":[],"neighbours":[{"slug":"2noise-chattts","name":"ChatTTS","tagline":"A generative speech model for daily dialogue","github_url":"https://github.com/2noise/ChatTTS","owner":"2noise","repo":"ChatTTS","owner_avatar_url":"https://avatars.githubusercontent.com/u/164844019?v=4","primary_language":"Python","stars":39768,"forks":4257,"topics":["agent","chat","chatgpt","chattts","chinese","chinese-language","english","english-language","gpt","llm","llm-agent","natural-language-inference","python","text-to-speech","torch","torchaudio","tts"],"archived":false,"github_pushed_at":"2026-04-10T16:33:48+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/2noise-chattts","markdown_url":"https://www.graphcanon.com/tools/2noise-chattts.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/2noise-chattts","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=2noise-chattts","shared_categories":["speech-audio"]},{"slug":"openbmb-voxcpm","name":"VoxCPM","tagline":"Tokenizer-Free TTS for Multilingual Speech Generation, Creative Voice Design, and True-to-Life Cloning","github_url":"https://github.com/OpenBMB/VoxCPM","owner":"OpenBMB","repo":"VoxCPM","owner_avatar_url":"https://avatars.githubusercontent.com/u/89920203?v=4","primary_language":"Python","stars":34452,"forks":3939,"topics":["audio","deeplearning","minicpm","multilingual","python","pytorch","speech","speech-synthesis","text-to-speech","tts","tts-model","voice-cloning","voice-design","voxcpm"],"archived":false,"github_pushed_at":"2026-07-08T09:46:11+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/openbmb-voxcpm","markdown_url":"https://www.graphcanon.com/tools/openbmb-voxcpm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openbmb-voxcpm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openbmb-voxcpm","shared_categories":["speech-audio"]},{"slug":"m-bain-whisperx","name":"whisperX","tagline":"WhisperX for automatic speech recognition with word-level timestamps and diarization","github_url":"https://github.com/m-bain/whisperX","owner":"m-bain","repo":"whisperX","owner_avatar_url":"https://avatars.githubusercontent.com/u/36994049?v=4","primary_language":"Python","stars":23329,"forks":2361,"topics":["asr","speech","speech-recognition","speech-to-text","whisper"],"archived":false,"github_pushed_at":"2026-07-13T08:30:07+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/m-bain-whisperx","markdown_url":"https://www.graphcanon.com/tools/m-bain-whisperx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/m-bain-whisperx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=m-bain-whisperx","shared_categories":["speech-audio"]},{"slug":"nari-labs-dia","name":"dia","tagline":"A TTS model for generating ultra-realistic dialogue","github_url":"https://github.com/nari-labs/dia","owner":"nari-labs","repo":"dia","owner_avatar_url":"https://avatars.githubusercontent.com/u/208232306?v=4","primary_language":"Python","stars":19361,"forks":1688,"topics":["ai","open-weight","text-to-speech"],"archived":false,"github_pushed_at":"2025-11-19T21:11:55+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/nari-labs-dia","markdown_url":"https://www.graphcanon.com/tools/nari-labs-dia.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nari-labs-dia","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nari-labs-dia","shared_categories":["speech-audio"]},{"slug":"aigc-audio-audiogpt","name":"AudioGPT","tagline":"AudioGPT: Understanding and Generating Speech, Music, Sound, and Talking Head","github_url":"https://github.com/AIGC-Audio/AudioGPT","owner":"AIGC-Audio","repo":"AudioGPT","owner_avatar_url":"https://avatars.githubusercontent.com/u/128504843?v=4","primary_language":"Python","stars":10172,"forks":850,"topics":["audio","gpt","music","sound","speech","talking-head"],"archived":false,"github_pushed_at":"2024-07-06T21:35:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/aigc-audio-audiogpt","markdown_url":"https://www.graphcanon.com/tools/aigc-audio-audiogpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/aigc-audio-audiogpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=aigc-audio-audiogpt","shared_categories":["speech-audio"]},{"slug":"netease-youdao-emotivoice","name":"EmotiVoice","tagline":"A Multi-Voice and Prompt-Controlled TTS Engine","github_url":"https://github.com/netease-youdao/EmotiVoice","owner":"netease-youdao","repo":"EmotiVoice","owner_avatar_url":"https://avatars.githubusercontent.com/u/3909232?v=4","primary_language":"Python","stars":8501,"forks":754,"topics":["ai","deep-learning","emotion","emotivoice","multi-speaker","prompt","python","pytorch","speech","speech-synthesis","style","text-to-speech","tts"],"archived":false,"github_pushed_at":"2024-08-13T10:23:52+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/netease-youdao-emotivoice","markdown_url":"https://www.graphcanon.com/tools/netease-youdao-emotivoice.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/netease-youdao-emotivoice","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=netease-youdao-emotivoice","shared_categories":["speech-audio"]},{"slug":"jaywalnut310-vits","name":"vits","tagline":"VITS: Conditional Variational Autoencoder with Adversarial Learning for End-to-End Text-to-Speech","github_url":"https://github.com/jaywalnut310/vits","owner":"jaywalnut310","repo":"vits","owner_avatar_url":"https://avatars.githubusercontent.com/u/20279210?v=4","primary_language":"Python","stars":7889,"forks":1384,"topics":["deep-learning","pytorch","speech-synthesis","text-to-speech","tts"],"archived":false,"github_pushed_at":"2023-12-06T01:29:50+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/jaywalnut310-vits","markdown_url":"https://www.graphcanon.com/tools/jaywalnut310-vits.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jaywalnut310-vits","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jaywalnut310-vits","shared_categories":["speech-audio"]},{"slug":"yl4579-styletts2","name":"StyleTTS2","tagline":"StyleTTS 2 advances human-like text-to-speech using style diffusion and adversarial training.","github_url":"https://github.com/yl4579/StyleTTS2","owner":"yl4579","repo":"StyleTTS2","owner_avatar_url":"https://avatars.githubusercontent.com/u/71044569?v=4","primary_language":"Python","stars":6322,"forks":694,"topics":["adversarial-training","deep-learning","diffusion-models","gan","latent-diffusion","latent-diffusion-models","pytorch","speaker-adaptation","speech-synthesis","text-to-speech","tts","wavlm"],"archived":false,"github_pushed_at":"2024-08-10T00:48:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/yl4579-styletts2","markdown_url":"https://www.graphcanon.com/tools/yl4579-styletts2.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/yl4579-styletts2","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=yl4579-styletts2","shared_categories":["speech-audio"]},{"slug":"snakers4-silero-models","name":"silero-models","tagline":"Silero Models provide simple access to pre-trained text-to-speech models","github_url":"https://github.com/snakers4/silero-models","owner":"snakers4","repo":"silero-models","owner_avatar_url":"https://avatars.githubusercontent.com/u/12515440?v=4","primary_language":"Jupyter Notebook","stars":6030,"forks":369,"topics":["armenian","azerbaijani","belarus","colab","georgian","kazakh","kyrgyz","pretrained-models","pytorch","russian","speech","speech-synthesis","speech-to-text","tajik","text-to-speech","torch-hub","tts","tts-models","ukrainian","uzbek"],"archived":false,"github_pushed_at":"2026-06-04T05:33:28+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/snakers4-silero-models","markdown_url":"https://www.graphcanon.com/tools/snakers4-silero-models.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/snakers4-silero-models","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=snakers4-silero-models","shared_categories":["speech-audio"]},{"slug":"metavoiceio-metavoice-src","name":"metavoice-src","tagline":"Foundational model for human-like, expressive TTS","github_url":"https://github.com/metavoiceio/metavoice-src","owner":"metavoiceio","repo":"metavoice-src","owner_avatar_url":"https://avatars.githubusercontent.com/u/107063843?v=4","primary_language":"Python","stars":4203,"forks":693,"topics":["ai","deep-learning","pytorch","speech","speech-synthesis","text-to-speech","tts","voice-clone","zero-shot-tts"],"archived":false,"github_pushed_at":"2024-07-30T22:13:01+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/metavoiceio-metavoice-src","markdown_url":"https://www.graphcanon.com/tools/metavoiceio-metavoice-src.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/metavoiceio-metavoice-src","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=metavoiceio-metavoice-src","shared_categories":["speech-audio"]},{"slug":"tensorspeech-tensorflowtts","name":"TensorFlowTTS","tagline":"Real-Time State-of-the-art Speech Synthesis for Tensorflow 2","github_url":"https://github.com/TensorSpeech/TensorFlowTTS","owner":"TensorSpeech","repo":"TensorFlowTTS","owner_avatar_url":"https://avatars.githubusercontent.com/u/67569298?v=4","primary_language":"Python","stars":3993,"forks":798,"topics":["chinese-tts","fastspeech","fastspeech2","german-tts","japanese-tts","korea-tts","melgan","mobile-tts","multi-speaker-tts","multiband-melgan","parallel-wavegan","real-time","speech-synthesis","tacotron2","tensorflow2","text-to-speech","tflite","tts","vocoder","zh-tts"],"archived":false,"github_pushed_at":"2024-07-05T07:24:49+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/tensorspeech-tensorflowtts","markdown_url":"https://www.graphcanon.com/tools/tensorspeech-tensorflowtts.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tensorspeech-tensorflowtts","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tensorspeech-tensorflowtts","shared_categories":["speech-audio"]},{"slug":"jik876-hifi-gan","name":"hifi-gan","tagline":"Generative Adversarial Networks for Efficient and High Fidelity Speech Synthesis","github_url":"https://github.com/jik876/hifi-gan","owner":"jik876","repo":"hifi-gan","owner_avatar_url":"https://avatars.githubusercontent.com/u/65878508?v=4","primary_language":"Python","stars":2363,"forks":555,"topics":["deep-learning","gan","hifi-gan","pytorch","speech-synthesis","text-to-speech","tts","vocoder"],"archived":false,"github_pushed_at":"2024-07-27T20:56:30+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/jik876-hifi-gan","markdown_url":"https://www.graphcanon.com/tools/jik876-hifi-gan.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jik876-hifi-gan","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jik876-hifi-gan","shared_categories":["speech-audio"]}]}}