{"data":{"node":{"slug":"jishengpeng-wavtokenizer","name":"WavTokenizer","tagline":"[ICLR 2025] State-of-the-art discrete acoustic codec models for audio language modeling","github_url":"https://github.com/jishengpeng/WavTokenizer","owner":"jishengpeng","repo":"WavTokenizer","owner_avatar_url":"https://avatars.githubusercontent.com/u/78149477?v=4","primary_language":"Python","stars":1310,"forks":113,"topics":["acoustic","audio-representation","codec","dac","encodec","gpt4o","music-representation-learning","semantic","soundstream","speech-language-model","speech-representation","text-to-speech"],"archived":false,"github_pushed_at":"2025-03-02T03:53:58+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/jishengpeng-wavtokenizer","markdown_url":"https://www.graphcanon.com/tools/jishengpeng-wavtokenizer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jishengpeng-wavtokenizer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jishengpeng-wavtokenizer"},"categories":[{"slug":"speech-audio","name":"Speech & Audio","url":"https://www.graphcanon.com/categories/speech-audio","markdown_url":"https://www.graphcanon.com/categories/speech-audio.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/speech-audio"}],"tags":[{"slug":"acoustic","name":"acoustic"},{"slug":"audio-representation","name":"audio-representation"},{"slug":"codec","name":"codec"},{"slug":"dac","name":"dac"},{"slug":"encodec","name":"encodec"},{"slug":"gpt4o","name":"gpt4o"},{"slug":"music-representation-learning","name":"music-representation-learning"},{"slug":"semantic","name":"semantic"}],"edges":[],"neighbours":[{"slug":"openbmb-voxcpm","name":"VoxCPM","tagline":"Tokenizer-Free TTS for Multilingual Speech Generation, Creative Voice Design, and True-to-Life Cloning","github_url":"https://github.com/OpenBMB/VoxCPM","owner":"OpenBMB","repo":"VoxCPM","owner_avatar_url":"https://avatars.githubusercontent.com/u/89920203?v=4","primary_language":"Python","stars":34452,"forks":3939,"topics":["audio","deeplearning","minicpm","multilingual","python","pytorch","speech","speech-synthesis","text-to-speech","tts","tts-model","voice-cloning","voice-design","voxcpm"],"archived":false,"github_pushed_at":"2026-07-08T09:46:11+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/openbmb-voxcpm","markdown_url":"https://www.graphcanon.com/tools/openbmb-voxcpm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openbmb-voxcpm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openbmb-voxcpm","shared_categories":["speech-audio"]},{"slug":"systran-faster-whisper","name":"faster-whisper","tagline":"Faster Whisper transcription with CTranslate2","github_url":"https://github.com/SYSTRAN/faster-whisper","owner":"SYSTRAN","repo":"faster-whisper","owner_avatar_url":"https://avatars.githubusercontent.com/u/1520500?v=4","primary_language":"Python","stars":24689,"forks":2006,"topics":["deep-learning","inference","openai","quantization","speech-recognition","speech-to-text","transformer","whisper"],"archived":false,"github_pushed_at":"2025-11-19T14:40:46+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/systran-faster-whisper","markdown_url":"https://www.graphcanon.com/tools/systran-faster-whisper.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/systran-faster-whisper","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=systran-faster-whisper","shared_categories":["speech-audio"]},{"slug":"m-bain-whisperx","name":"whisperX","tagline":"WhisperX for automatic speech recognition with word-level timestamps and diarization","github_url":"https://github.com/m-bain/whisperX","owner":"m-bain","repo":"whisperX","owner_avatar_url":"https://avatars.githubusercontent.com/u/36994049?v=4","primary_language":"Python","stars":23329,"forks":2361,"topics":["asr","speech","speech-recognition","speech-to-text","whisper"],"archived":false,"github_pushed_at":"2026-07-13T08:30:07+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/m-bain-whisperx","markdown_url":"https://www.graphcanon.com/tools/m-bain-whisperx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/m-bain-whisperx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=m-bain-whisperx","shared_categories":["speech-audio"]},{"slug":"speechbrain-speechbrain","name":"speechbrain","tagline":"A PyTorch-based Speech Toolkit","github_url":"https://github.com/speechbrain/speechbrain","owner":"speechbrain","repo":"speechbrain","owner_avatar_url":"https://avatars.githubusercontent.com/u/54749030?v=4","primary_language":"Python","stars":11725,"forks":1712,"topics":["asr","audio","audio-processing","deep-learning","huggingface","language-model","pytorch","speaker-diarization","speaker-recognition","speaker-verification","speech-enhancement","speech-processing","speech-recognition","speech-separation","speech-to-text","speech-toolkit","speechrecognition","spoken-language-understanding","transformers","voice-recognition"],"archived":false,"github_pushed_at":"2026-06-15T11:24:25+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/speechbrain-speechbrain","markdown_url":"https://www.graphcanon.com/tools/speechbrain-speechbrain.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/speechbrain-speechbrain","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=speechbrain-speechbrain","shared_categories":["speech-audio"]},{"slug":"huggingface-tokenizers","name":"tokenizers","tagline":"💥 Fast State-of-the-Art Tokenizers optimized for Research and Production","github_url":"https://github.com/huggingface/tokenizers","owner":"huggingface","repo":"tokenizers","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Rust","stars":10940,"forks":1160,"topics":["bert","gpt","language-model","natural-language-processing","natural-language-understanding","nlp","transformers"],"archived":false,"github_pushed_at":"2026-08-01T12:35:36+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-tokenizers","markdown_url":"https://www.graphcanon.com/tools/huggingface-tokenizers.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-tokenizers","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-tokenizers","shared_categories":[]},{"slug":"espnet-espnet","name":"espnet","tagline":"End-to-End Speech Processing Toolkit","github_url":"https://github.com/espnet/espnet","owner":"espnet","repo":"espnet","owner_avatar_url":"https://avatars.githubusercontent.com/u/34493687?v=4","primary_language":"Python","stars":9903,"forks":2421,"topics":["chainer","deep-learning","end-to-end","kaldi","machine-translation","pytorch","singing-voice-synthesis","speaker-diarization","speech-enhancement","speech-recognition","speech-separation","speech-synthesis","speech-translation","spoken-language-understanding","text-to-speech","voice-conversion"],"archived":false,"github_pushed_at":"2026-07-28T14:36:55+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/espnet-espnet","markdown_url":"https://www.graphcanon.com/tools/espnet-espnet.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/espnet-espnet","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=espnet-espnet","shared_categories":["speech-audio"]},{"slug":"jaywalnut310-vits","name":"vits","tagline":"VITS: Conditional Variational Autoencoder with Adversarial Learning for End-to-End Text-to-Speech","github_url":"https://github.com/jaywalnut310/vits","owner":"jaywalnut310","repo":"vits","owner_avatar_url":"https://avatars.githubusercontent.com/u/20279210?v=4","primary_language":"Python","stars":7889,"forks":1384,"topics":["deep-learning","pytorch","speech-synthesis","text-to-speech","tts"],"archived":false,"github_pushed_at":"2023-12-06T01:29:50+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/jaywalnut310-vits","markdown_url":"https://www.graphcanon.com/tools/jaywalnut310-vits.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jaywalnut310-vits","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jaywalnut310-vits","shared_categories":["speech-audio"]},{"slug":"yl4579-styletts2","name":"StyleTTS2","tagline":"StyleTTS 2 advances human-like text-to-speech using style diffusion and adversarial training.","github_url":"https://github.com/yl4579/StyleTTS2","owner":"yl4579","repo":"StyleTTS2","owner_avatar_url":"https://avatars.githubusercontent.com/u/71044569?v=4","primary_language":"Python","stars":6322,"forks":694,"topics":["adversarial-training","deep-learning","diffusion-models","gan","latent-diffusion","latent-diffusion-models","pytorch","speaker-adaptation","speech-synthesis","text-to-speech","tts","wavlm"],"archived":false,"github_pushed_at":"2024-08-10T00:48:18+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/yl4579-styletts2","markdown_url":"https://www.graphcanon.com/tools/yl4579-styletts2.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/yl4579-styletts2","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=yl4579-styletts2","shared_categories":["speech-audio"]},{"slug":"snakers4-silero-models","name":"silero-models","tagline":"Silero Models provide simple access to pre-trained text-to-speech models","github_url":"https://github.com/snakers4/silero-models","owner":"snakers4","repo":"silero-models","owner_avatar_url":"https://avatars.githubusercontent.com/u/12515440?v=4","primary_language":"Jupyter Notebook","stars":6030,"forks":369,"topics":["armenian","azerbaijani","belarus","colab","georgian","kazakh","kyrgyz","pretrained-models","pytorch","russian","speech","speech-synthesis","speech-to-text","tajik","text-to-speech","torch-hub","tts","tts-models","ukrainian","uzbek"],"archived":false,"github_pushed_at":"2026-06-04T05:33:28+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/snakers4-silero-models","markdown_url":"https://www.graphcanon.com/tools/snakers4-silero-models.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/snakers4-silero-models","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=snakers4-silero-models","shared_categories":["speech-audio"]},{"slug":"meta-pytorch-torchtune","name":"torchtune","tagline":"PyTorch native post-training library","github_url":"https://github.com/meta-pytorch/torchtune","owner":"meta-pytorch","repo":"torchtune","owner_avatar_url":"https://avatars.githubusercontent.com/u/107212512?v=4","primary_language":"Python","stars":5793,"forks":743,"topics":[],"archived":false,"github_pushed_at":"2026-08-06T12:15:22+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/meta-pytorch-torchtune","markdown_url":"https://www.graphcanon.com/tools/meta-pytorch-torchtune.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/meta-pytorch-torchtune","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=meta-pytorch-torchtune","shared_categories":[]},{"slug":"sanchit-gandhi-whisper-jax","name":"whisper-jax","tagline":"JAX implementation of OpenAI's Whisper model for up to 70x speed-up on TPU.","github_url":"https://github.com/sanchit-gandhi/whisper-jax","owner":"sanchit-gandhi","repo":"whisper-jax","owner_avatar_url":"https://avatars.githubusercontent.com/u/93869735?v=4","primary_language":"Jupyter Notebook","stars":4684,"forks":411,"topics":["deep-learning","jax","speech-recognition","speech-to-text","whisper"],"archived":false,"github_pushed_at":"2024-04-03T12:12:52+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/sanchit-gandhi-whisper-jax","markdown_url":"https://www.graphcanon.com/tools/sanchit-gandhi-whisper-jax.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/sanchit-gandhi-whisper-jax","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=sanchit-gandhi-whisper-jax","shared_categories":["speech-audio"]},{"slug":"metavoiceio-metavoice-src","name":"metavoice-src","tagline":"Foundational model for human-like, expressive TTS","github_url":"https://github.com/metavoiceio/metavoice-src","owner":"metavoiceio","repo":"metavoice-src","owner_avatar_url":"https://avatars.githubusercontent.com/u/107063843?v=4","primary_language":"Python","stars":4203,"forks":693,"topics":["ai","deep-learning","pytorch","speech","speech-synthesis","text-to-speech","tts","voice-clone","zero-shot-tts"],"archived":false,"github_pushed_at":"2024-07-30T22:13:01+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/metavoiceio-metavoice-src","markdown_url":"https://www.graphcanon.com/tools/metavoiceio-metavoice-src.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/metavoiceio-metavoice-src","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=metavoiceio-metavoice-src","shared_categories":["speech-audio"]}]}}