{"data":{"node":{"slug":"coqui-ai-stt","name":"STT","tagline":"A fast open-source deep-learning toolkit for speech-to-text","github_url":"https://github.com/coqui-ai/STT","owner":"coqui-ai","repo":"STT","owner_avatar_url":"https://avatars.githubusercontent.com/u/75583352?v=4","primary_language":"C++","stars":2599,"forks":299,"topics":["asr","automatic-speech-recognition","deep-learning","speech-recognition","speech-recognition-api","speech-recognizer","speech-to-text","stt","tensorflow","voice-recognition"],"archived":false,"github_pushed_at":"2024-03-11T07:55:56+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/coqui-ai-stt","markdown_url":"https://www.graphcanon.com/tools/coqui-ai-stt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/coqui-ai-stt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=coqui-ai-stt"},"categories":[{"slug":"speech-audio","name":"Speech & Audio","url":"https://www.graphcanon.com/categories/speech-audio","markdown_url":"https://www.graphcanon.com/categories/speech-audio.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/speech-audio"}],"tags":[{"slug":"asr","name":"asr"},{"slug":"automatic-speech-recognition","name":"automatic-speech-recognition"},{"slug":"deep-learning","name":"deep-learning"},{"slug":"speech-recognition","name":"speech-recognition"},{"slug":"tensorflow","name":"tensorflow"}],"edges":[],"neighbours":[{"slug":"coqui-ai-tts","name":"TTS","tagline":"🐸💬 - a deep learning toolkit for Text-to-Speech","github_url":"https://github.com/coqui-ai/TTS","owner":"coqui-ai","repo":"TTS","owner_avatar_url":"https://avatars.githubusercontent.com/u/75583352?v=4","primary_language":"Python","stars":45832,"forks":6158,"topics":["deep-learning","glow-tts","hifigan","melgan","multi-speaker-tts","python","pytorch","speaker-encoder","speaker-encodings","speech","speech-synthesis","tacotron","text-to-speech","tts","tts-model","vocoder","voice-cloning","voice-conversion","voice-synthesis"],"archived":false,"github_pushed_at":"2024-08-16T12:07:14+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/coqui-ai-tts","markdown_url":"https://www.graphcanon.com/tools/coqui-ai-tts.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/coqui-ai-tts","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=coqui-ai-tts","shared_categories":["speech-audio"]},{"slug":"nvidia-nemo-speech","name":"Speech","tagline":"A scalable generative AI framework for Speech AI","github_url":"https://github.com/NVIDIA-NeMo/Speech","owner":"NVIDIA-NeMo","repo":"Speech","owner_avatar_url":"https://avatars.githubusercontent.com/u/213689629?v=4","primary_language":"Python","stars":17940,"forks":3533,"topics":["asr","deeplearning","generative-ai","machine-translation","neural-networks","speaker-diariazation","speaker-recognition","speech-synthesis","speech-translation","tts"],"archived":false,"github_pushed_at":"2026-08-07T05:34:13+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/nvidia-nemo-speech","markdown_url":"https://www.graphcanon.com/tools/nvidia-nemo-speech.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-nemo-speech","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-nemo-speech","shared_categories":["speech-audio"]},{"slug":"speechbrain-speechbrain","name":"speechbrain","tagline":"A PyTorch-based Speech Toolkit","github_url":"https://github.com/speechbrain/speechbrain","owner":"speechbrain","repo":"speechbrain","owner_avatar_url":"https://avatars.githubusercontent.com/u/54749030?v=4","primary_language":"Python","stars":11725,"forks":1712,"topics":["asr","audio","audio-processing","deep-learning","huggingface","language-model","pytorch","speaker-diarization","speaker-recognition","speaker-verification","speech-enhancement","speech-processing","speech-recognition","speech-separation","speech-to-text","speech-toolkit","speechrecognition","spoken-language-understanding","transformers","voice-recognition"],"archived":false,"github_pushed_at":"2026-06-15T11:24:25+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/speechbrain-speechbrain","markdown_url":"https://www.graphcanon.com/tools/speechbrain-speechbrain.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/speechbrain-speechbrain","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=speechbrain-speechbrain","shared_categories":["speech-audio"]},{"slug":"espnet-espnet","name":"espnet","tagline":"End-to-End Speech Processing Toolkit","github_url":"https://github.com/espnet/espnet","owner":"espnet","repo":"espnet","owner_avatar_url":"https://avatars.githubusercontent.com/u/34493687?v=4","primary_language":"Python","stars":9903,"forks":2421,"topics":["chainer","deep-learning","end-to-end","kaldi","machine-translation","pytorch","singing-voice-synthesis","speaker-diarization","speech-enhancement","speech-recognition","speech-separation","speech-synthesis","speech-translation","spoken-language-understanding","text-to-speech","voice-conversion"],"archived":false,"github_pushed_at":"2026-07-28T14:36:55+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/espnet-espnet","markdown_url":"https://www.graphcanon.com/tools/espnet-espnet.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/espnet-espnet","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=espnet-espnet","shared_categories":["speech-audio"]},{"slug":"huggingface-speech-to-speech","name":"speech-to-speech","tagline":"Build local voice agents with open-source models","github_url":"https://github.com/huggingface/speech-to-speech","owner":"huggingface","repo":"speech-to-speech","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":8219,"forks":1025,"topics":["ai","assistant","language-model","machine-learning","python","speech","speech-synthesis","speech-to-text","speech-translation"],"archived":false,"github_pushed_at":"2026-07-30T11:35:50+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/huggingface-speech-to-speech","markdown_url":"https://www.graphcanon.com/tools/huggingface-speech-to-speech.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-speech-to-speech","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-speech-to-speech","shared_categories":["speech-audio"]},{"slug":"blaizzy-mlx-audio","name":"mlx-audio","tagline":"A text-to-speech (TTS), speech-to-text (STT) and speech-to-speech (STS) library on Apple's MLX framework.","github_url":"https://github.com/Blaizzy/mlx-audio","owner":"Blaizzy","repo":"mlx-audio","owner_avatar_url":"https://avatars.githubusercontent.com/u/23445657?v=4","primary_language":"Python","stars":7639,"forks":680,"topics":["apple-silicon","audio-processing","mlx","multimodal","speech-recognition","speech-synthesis","speech-to-text","text-to-speech","transformers"],"archived":false,"github_pushed_at":"2026-07-28T16:14:36+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/blaizzy-mlx-audio","markdown_url":"https://www.graphcanon.com/tools/blaizzy-mlx-audio.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/blaizzy-mlx-audio","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=blaizzy-mlx-audio","shared_categories":["speech-audio"]},{"slug":"remsky-kokoro-fastapi","name":"Kokoro-FastAPI","tagline":"Dockerized FastAPI wrapper for Kokoro-82M text-to-speech model","github_url":"https://github.com/remsky/Kokoro-FastAPI","owner":"remsky","repo":"Kokoro-FastAPI","owner_avatar_url":"https://avatars.githubusercontent.com/u/25017870?v=4","primary_language":"Python","stars":5265,"forks":858,"topics":["fastapi","huggingface-spaces","kokoro","kokoro-tts","openai-compatible-api","openwebui","pytorch","sillytavern","text-to-speech","tts","tts-api","uv"],"archived":false,"github_pushed_at":"2026-07-21T03:16:48+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/remsky-kokoro-fastapi","markdown_url":"https://www.graphcanon.com/tools/remsky-kokoro-fastapi.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/remsky-kokoro-fastapi","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=remsky-kokoro-fastapi","shared_categories":["speech-audio"]},{"slug":"jianchang512-stt","name":"stt","tagline":"A local offline speech-to-text tool for video and audio files.","github_url":"https://github.com/jianchang512/stt","owner":"jianchang512","repo":"stt","owner_avatar_url":"https://avatars.githubusercontent.com/u/3378335?v=4","primary_language":"Python","stars":4712,"forks":493,"topics":["speech","speech-recognition","speech-to-text","stt"],"archived":false,"github_pushed_at":"2026-01-22T08:38:53+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/jianchang512-stt","markdown_url":"https://www.graphcanon.com/tools/jianchang512-stt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jianchang512-stt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jianchang512-stt","shared_categories":["speech-audio"]},{"slug":"tensorspeech-tensorflowtts","name":"TensorFlowTTS","tagline":"Real-Time State-of-the-art Speech Synthesis for Tensorflow 2","github_url":"https://github.com/TensorSpeech/TensorFlowTTS","owner":"TensorSpeech","repo":"TensorFlowTTS","owner_avatar_url":"https://avatars.githubusercontent.com/u/67569298?v=4","primary_language":"Python","stars":3993,"forks":798,"topics":["chinese-tts","fastspeech","fastspeech2","german-tts","japanese-tts","korea-tts","melgan","mobile-tts","multi-speaker-tts","multiband-melgan","parallel-wavegan","real-time","speech-synthesis","tacotron2","tensorflow2","text-to-speech","tflite","tts","vocoder","zh-tts"],"archived":false,"github_pushed_at":"2024-07-05T07:24:49+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/tensorspeech-tensorflowtts","markdown_url":"https://www.graphcanon.com/tools/tensorspeech-tensorflowtts.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/tensorspeech-tensorflowtts","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=tensorspeech-tensorflowtts","shared_categories":["speech-audio"]},{"slug":"pannous-tensorflow-speech-recognition","name":"tensorflow-speech-recognition","tagline":"Speech recognition using TensorFlow deep learning framework","github_url":"https://github.com/pannous/tensorflow-speech-recognition","owner":"pannous","repo":"tensorflow-speech-recognition","owner_avatar_url":"https://avatars.githubusercontent.com/u/516118?v=4","primary_language":"Python","stars":2173,"forks":631,"topics":["deep-learning","neural-network","speech-recognition","speech-to-text","stt","tensorflow"],"archived":false,"github_pushed_at":"2024-01-17T14:27:13+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/pannous-tensorflow-speech-recognition","markdown_url":"https://www.graphcanon.com/tools/pannous-tensorflow-speech-recognition.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/pannous-tensorflow-speech-recognition","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=pannous-tensorflow-speech-recognition","shared_categories":["speech-audio"]},{"slug":"ictnlp-streamspeech","name":"StreamSpeech","tagline":"All-in-one speech recognition and synthesis model for offline and simultaneous processing","github_url":"https://github.com/ictnlp/StreamSpeech","owner":"ictnlp","repo":"StreamSpeech","owner_avatar_url":"https://avatars.githubusercontent.com/u/45630465?v=4","primary_language":"Python","stars":1278,"forks":103,"topics":["all-in-one","asr","audio-processing","machine-translation","non-autoregressive","seamless","simultaneous-translation","speech","speech-enhancement","speech-processing","speech-recognition","speech-synthesis","speech-to-text","speech-translation","streaming-audio","text-to-audio","text-to-speech","translation","tts","voice"],"archived":false,"github_pushed_at":"2025-06-29T02:06:27+00:00","maintenance_label":"Dormant","url":"https://www.graphcanon.com/tools/ictnlp-streamspeech","markdown_url":"https://www.graphcanon.com/tools/ictnlp-streamspeech.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ictnlp-streamspeech","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ictnlp-streamspeech","shared_categories":["speech-audio"]}]}}