{"data":{"node":{"slug":"rayhane-mamah-tacotron-2","name":"Tacotron-2","tagline":"DeepMind's Tacotron-2 Tensorflow implementation focused on text-to-speech synthesis.","github_url":"https://github.com/Rayhane-mamah/Tacotron-2","owner":"Rayhane-mamah","repo":"Tacotron-2","owner_avatar_url":"https://avatars.githubusercontent.com/u/34689728?v=4","primary_language":"Python","stars":2324,"forks":899,"topics":["paper","python","speech-synthesis","tacotron","tensorflow","text-to-speech","wavenet"],"archived":false,"github_pushed_at":"2023-07-06T21:18:43+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/rayhane-mamah-tacotron-2","markdown_url":"https://www.graphcanon.com/tools/rayhane-mamah-tacotron-2.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rayhane-mamah-tacotron-2","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rayhane-mamah-tacotron-2"},"categories":[{"slug":"speech-audio","name":"Speech & Audio","url":"https://www.graphcanon.com/categories/speech-audio","markdown_url":"https://www.graphcanon.com/categories/speech-audio.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/speech-audio"}],"tags":[{"slug":"paper","name":"paper"},{"slug":"python","name":"python"},{"slug":"speech-synthesis","name":"speech-synthesis"},{"slug":"tacotron","name":"tacotron"},{"slug":"tensorflow","name":"tensorflow"},{"slug":"text-to-speech","name":"text-to-speech"},{"slug":"wavenet","name":"wavenet"}],"edges":[],"neighbours":[{"slug":"tensorflow-tensorflow","name":"tensorflow","tagline":"An Open Source Machine Learning Framework for Everyone","github_url":"https://github.com/tensorflow/tensorflow","owner":"tensorflow","repo":"tensorflow","owner_avatar_url":"https://avatars.githubusercontent.com/u/15658638?v=4","primary_language":"C++","stars":196758,"forks":75773,"topics":["deep-learning","deep-neural-networks","distributed","machine-learning","ml","neural-network","python","tensorflow"],"archived":false,"github_pushed_at":"2026-08-03T06:01:06+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/tensorflow-tensorflow","markdown_url":"https://www.graphcanon.com/tools/tensorflow-tensorflow.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tensorflow-tensorflow","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tensorflow-tensorflow","shared_categories":[]},{"slug":"coqui-ai-tts","name":"TTS","tagline":"🐸💬 - a deep learning toolkit for Text-to-Speech","github_url":"https://github.com/coqui-ai/TTS","owner":"coqui-ai","repo":"TTS","owner_avatar_url":"https://avatars.githubusercontent.com/u/75583352?v=4","primary_language":"Python","stars":45832,"forks":6158,"topics":["deep-learning","glow-tts","hifigan","melgan","multi-speaker-tts","python","pytorch","speaker-encoder","speaker-encodings","speech","speech-synthesis","tacotron","text-to-speech","tts","tts-model","vocoder","voice-cloning","voice-conversion","voice-synthesis"],"archived":false,"github_pushed_at":"2024-08-16T12:07:14+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/coqui-ai-tts","markdown_url":"https://www.graphcanon.com/tools/coqui-ai-tts.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/coqui-ai-tts","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=coqui-ai-tts","shared_categories":["speech-audio"]},{"slug":"openbmb-voxcpm","name":"VoxCPM","tagline":"Tokenizer-Free TTS for Multilingual Speech Generation, Creative Voice Design, and True-to-Life Cloning","github_url":"https://github.com/OpenBMB/VoxCPM","owner":"OpenBMB","repo":"VoxCPM","owner_avatar_url":"https://avatars.githubusercontent.com/u/89920203?v=4","primary_language":"Python","stars":34452,"forks":3939,"topics":["audio","deeplearning","minicpm","multilingual","python","pytorch","speech","speech-synthesis","text-to-speech","tts","tts-model","voice-cloning","voice-design","voxcpm"],"archived":false,"github_pushed_at":"2026-07-08T09:46:11+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/openbmb-voxcpm","markdown_url":"https://www.graphcanon.com/tools/openbmb-voxcpm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openbmb-voxcpm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openbmb-voxcpm","shared_categories":["speech-audio"]},{"slug":"systran-faster-whisper","name":"faster-whisper","tagline":"Faster Whisper transcription with CTranslate2","github_url":"https://github.com/SYSTRAN/faster-whisper","owner":"SYSTRAN","repo":"faster-whisper","owner_avatar_url":"https://avatars.githubusercontent.com/u/1520500?v=4","primary_language":"Python","stars":24689,"forks":2006,"topics":["deep-learning","inference","openai","quantization","speech-recognition","speech-to-text","transformer","whisper"],"archived":false,"github_pushed_at":"2025-11-19T14:40:46+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/systran-faster-whisper","markdown_url":"https://www.graphcanon.com/tools/systran-faster-whisper.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/systran-faster-whisper","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=systran-faster-whisper","shared_categories":["speech-audio"]},{"slug":"nari-labs-dia","name":"dia","tagline":"A TTS model for generating ultra-realistic dialogue","github_url":"https://github.com/nari-labs/dia","owner":"nari-labs","repo":"dia","owner_avatar_url":"https://avatars.githubusercontent.com/u/208232306?v=4","primary_language":"Python","stars":19361,"forks":1688,"topics":["ai","open-weight","text-to-speech"],"archived":false,"github_pushed_at":"2025-11-19T21:11:55+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/nari-labs-dia","markdown_url":"https://www.graphcanon.com/tools/nari-labs-dia.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nari-labs-dia","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nari-labs-dia","shared_categories":["speech-audio"]},{"slug":"supertone-inc-supertonic","name":"supertonic","tagline":"Lightning-Fast On-Device Multilingual TTS via ONNX","github_url":"https://github.com/supertone-inc/supertonic","owner":"supertone-inc","repo":"supertonic","owner_avatar_url":"https://avatars.githubusercontent.com/u/61007505?v=4","primary_language":"Swift","stars":13543,"forks":1427,"topics":["cpp","csharp","flutter","go","ios","java","lightweight","multilingual","nodejs","on-device","onnx","onnxruntime","python","rust","speech-synthesis","swift","text-to-speech","tts","web","webgpu"],"archived":false,"github_pushed_at":"2026-07-24T04:00:17+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/supertone-inc-supertonic","markdown_url":"https://www.graphcanon.com/tools/supertone-inc-supertonic.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/supertone-inc-supertonic","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=supertone-inc-supertonic","shared_categories":["speech-audio"]},{"slug":"speechbrain-speechbrain","name":"speechbrain","tagline":"A PyTorch-based Speech Toolkit","github_url":"https://github.com/speechbrain/speechbrain","owner":"speechbrain","repo":"speechbrain","owner_avatar_url":"https://avatars.githubusercontent.com/u/54749030?v=4","primary_language":"Python","stars":11725,"forks":1712,"topics":["asr","audio","audio-processing","deep-learning","huggingface","language-model","pytorch","speaker-diarization","speaker-recognition","speaker-verification","speech-enhancement","speech-processing","speech-recognition","speech-separation","speech-to-text","speech-toolkit","speechrecognition","spoken-language-understanding","transformers","voice-recognition"],"archived":false,"github_pushed_at":"2026-06-15T11:24:25+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/speechbrain-speechbrain","markdown_url":"https://www.graphcanon.com/tools/speechbrain-speechbrain.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/speechbrain-speechbrain","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=speechbrain-speechbrain","shared_categories":["speech-audio"]},{"slug":"espnet-espnet","name":"espnet","tagline":"End-to-End Speech Processing Toolkit","github_url":"https://github.com/espnet/espnet","owner":"espnet","repo":"espnet","owner_avatar_url":"https://avatars.githubusercontent.com/u/34493687?v=4","primary_language":"Python","stars":9903,"forks":2421,"topics":["chainer","deep-learning","end-to-end","kaldi","machine-translation","pytorch","singing-voice-synthesis","speaker-diarization","speech-enhancement","speech-recognition","speech-separation","speech-synthesis","speech-translation","spoken-language-understanding","text-to-speech","voice-conversion"],"archived":false,"github_pushed_at":"2026-07-28T14:36:55+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/espnet-espnet","markdown_url":"https://www.graphcanon.com/tools/espnet-espnet.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/espnet-espnet","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=espnet-espnet","shared_categories":["speech-audio"]},{"slug":"jaywalnut310-vits","name":"vits","tagline":"VITS: Conditional Variational Autoencoder with Adversarial Learning for End-to-End Text-to-Speech","github_url":"https://github.com/jaywalnut310/vits","owner":"jaywalnut310","repo":"vits","owner_avatar_url":"https://avatars.githubusercontent.com/u/20279210?v=4","primary_language":"Python","stars":7889,"forks":1384,"topics":["deep-learning","pytorch","speech-synthesis","text-to-speech","tts"],"archived":false,"github_pushed_at":"2023-12-06T01:29:50+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/jaywalnut310-vits","markdown_url":"https://www.graphcanon.com/tools/jaywalnut310-vits.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jaywalnut310-vits","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jaywalnut310-vits","shared_categories":["speech-audio"]},{"slug":"snakers4-silero-models","name":"silero-models","tagline":"Silero Models provide simple access to pre-trained text-to-speech models","github_url":"https://github.com/snakers4/silero-models","owner":"snakers4","repo":"silero-models","owner_avatar_url":"https://avatars.githubusercontent.com/u/12515440?v=4","primary_language":"Jupyter Notebook","stars":6030,"forks":369,"topics":["armenian","azerbaijani","belarus","colab","georgian","kazakh","kyrgyz","pretrained-models","pytorch","russian","speech","speech-synthesis","speech-to-text","tajik","text-to-speech","torch-hub","tts","tts-models","ukrainian","uzbek"],"archived":false,"github_pushed_at":"2026-06-04T05:33:28+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/snakers4-silero-models","markdown_url":"https://www.graphcanon.com/tools/snakers4-silero-models.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/snakers4-silero-models","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=snakers4-silero-models","shared_categories":["speech-audio"]},{"slug":"sanchit-gandhi-whisper-jax","name":"whisper-jax","tagline":"JAX implementation of OpenAI's Whisper model for up to 70x speed-up on TPU.","github_url":"https://github.com/sanchit-gandhi/whisper-jax","owner":"sanchit-gandhi","repo":"whisper-jax","owner_avatar_url":"https://avatars.githubusercontent.com/u/93869735?v=4","primary_language":"Jupyter Notebook","stars":4684,"forks":411,"topics":["deep-learning","jax","speech-recognition","speech-to-text","whisper"],"archived":false,"github_pushed_at":"2024-04-03T12:12:52+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/sanchit-gandhi-whisper-jax","markdown_url":"https://www.graphcanon.com/tools/sanchit-gandhi-whisper-jax.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/sanchit-gandhi-whisper-jax","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=sanchit-gandhi-whisper-jax","shared_categories":["speech-audio"]},{"slug":"metavoiceio-metavoice-src","name":"metavoice-src","tagline":"Foundational model for human-like, expressive TTS","github_url":"https://github.com/metavoiceio/metavoice-src","owner":"metavoiceio","repo":"metavoice-src","owner_avatar_url":"https://avatars.githubusercontent.com/u/107063843?v=4","primary_language":"Python","stars":4203,"forks":693,"topics":["ai","deep-learning","pytorch","speech","speech-synthesis","text-to-speech","tts","voice-clone","zero-shot-tts"],"archived":false,"github_pushed_at":"2024-07-30T22:13:01+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/metavoiceio-metavoice-src","markdown_url":"https://www.graphcanon.com/tools/metavoiceio-metavoice-src.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/metavoiceio-metavoice-src","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=metavoiceio-metavoice-src","shared_categories":["speech-audio"]}]}}