{"data":{"slug":"speech-audio","name":"Speech & Audio","url":"https://www.graphcanon.com/categories/speech-audio","markdown_url":"https://www.graphcanon.com/categories/speech-audio.md","api_url":"https://www.graphcanon.com/api/graphcanon/categories/speech-audio","tool_count":205,"tools":[{"slug":"huggingface-transformers","name":"transformers","tagline":"Transformers: the model-definition framework for state-of-the-art machine learning models in text, vision, audio, and multimodal models","github_url":"https://github.com/huggingface/transformers","owner":"huggingface","repo":"transformers","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":164121,"forks":34249,"topics":["audio","deep-learning","deepseek","gemma","glm","hacktoberfest","llm","machine-learning","model-hub","natural-language-processing","nlp","pretrained-models","python","pytorch","pytorch-transformers","qwen","speech-recognition","transformer","vlm"],"archived":false,"github_pushed_at":"2026-08-15T22:28:12+00:00","maintenance_label":"Very active","stars_delta_30d":1457,"url":"https://www.graphcanon.com/tools/huggingface-transformers","markdown_url":"https://www.graphcanon.com/tools/huggingface-transformers.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-transformers","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-transformers"},{"slug":"mudler-localai","name":"LocalAI","tagline":"Run any model - LLMs, vision, voice, image, video - on any hardware. No GPU required.","github_url":"https://github.com/mudler/LocalAI","owner":"mudler","repo":"LocalAI","owner_avatar_url":"https://avatars.githubusercontent.com/u/2420543?v=4","primary_language":"Go","stars":48500,"forks":4362,"topics":["agents","ai","api","audio-generation","decentralized","distributed","image-generation","libp2p","llama","llm","mamba","mcp","musicgen","object-detection","rerank","stable-diffusion","text-generation","tts"],"archived":false,"github_pushed_at":"2026-08-16T05:07:25+00:00","maintenance_label":"Very active","stars_delta_30d":924,"url":"https://www.graphcanon.com/tools/mudler-localai","markdown_url":"https://www.graphcanon.com/tools/mudler-localai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/mudler-localai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=mudler-localai"},{"slug":"ggml-org-whisper-cpp","name":"whisper.cpp","tagline":"Port of OpenAI's Whisper model in C/C++ for speech-to-text inference","github_url":"https://github.com/ggml-org/whisper.cpp","owner":"ggml-org","repo":"whisper.cpp","owner_avatar_url":"https://avatars.githubusercontent.com/u/134263123?v=4","primary_language":"C++","stars":52501,"forks":5971,"topics":["inference","openai","speech-recognition","speech-to-text","transformer","whisper"],"archived":false,"github_pushed_at":"2026-07-31T07:11:28+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/ggml-org-whisper-cpp","markdown_url":"https://www.graphcanon.com/tools/ggml-org-whisper-cpp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ggml-org-whisper-cpp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ggml-org-whisper-cpp"},{"slug":"screenpipe-screenpipe","name":"screenpipe","tagline":"AI that records and analyzes everything you do, say, hear locally","github_url":"https://github.com/screenpipe/screenpipe","owner":"screenpipe","repo":"screenpipe","owner_avatar_url":"https://avatars.githubusercontent.com/u/259178917?v=4","primary_language":"Rust","stars":20534,"forks":2029,"topics":["agents","agi","ai","ai-memory","audio-recording","computer-vision","hermes","hermes-agent","llm","local-ai","local-first","machine-learning","mcp","multimodal","openclaw","privacy","rewind","screen-recording","speech-to-text","ycombinator"],"archived":false,"github_pushed_at":"2026-07-26T03:40:56+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/screenpipe-screenpipe","markdown_url":"https://www.graphcanon.com/tools/screenpipe-screenpipe.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/screenpipe-screenpipe","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=screenpipe-screenpipe"},{"slug":"modelscope-funasr","name":"FunASR","tagline":"Industrial-grade speech recognition toolkit","github_url":"https://github.com/modelscope/FunASR","owner":"modelscope","repo":"FunASR","owner_avatar_url":"https://avatars.githubusercontent.com/u/109945100?v=4","primary_language":"Python","stars":19554,"forks":1965,"topics":["asr","audio","chinese","emotion-recognition","funasr","mcp-server","multilingual-asr","openai-compatible-api","paraformer","punctuation","pytorch","real-time-asr","speaker-diarization","speech-recognition","speech-to-text","streaming-asr","transcription","vllm","voice-activity-detection","whisper-alternative"],"archived":false,"github_pushed_at":"2026-07-30T01:45:38+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/modelscope-funasr","markdown_url":"https://www.graphcanon.com/tools/modelscope-funasr.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/modelscope-funasr","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=modelscope-funasr"},{"slug":"mastra-ai-mastra","name":"mastra","tagline":"Modern TypeScript framework for AI-powered applications and agents","github_url":"https://github.com/mastra-ai/mastra","owner":"mastra-ai","repo":"mastra","owner_avatar_url":"https://avatars.githubusercontent.com/u/149120496?v=4","primary_language":"TypeScript","stars":27229,"forks":2642,"topics":["agents","ai","chatbots","evals","javascript","llm","mcp","nextjs","nodejs","reactjs","tts","typescript","workflows"],"archived":false,"github_pushed_at":"2026-08-16T17:54:58+00:00","maintenance_label":"Very active","stars_delta_30d":954,"url":"https://www.graphcanon.com/tools/mastra-ai-mastra","markdown_url":"https://www.graphcanon.com/tools/mastra-ai-mastra.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/mastra-ai-mastra","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=mastra-ai-mastra"},{"slug":"funaudiollm-cosyvoice","name":"CosyVoice","tagline":"Multi-lingual large voice generation model with full-stack abilities for inference, training and deployment.","github_url":"https://github.com/FunAudioLLM/CosyVoice","owner":"FunAudioLLM","repo":"CosyVoice","owner_avatar_url":"https://avatars.githubusercontent.com/u/167062371?v=4","primary_language":"Python","stars":22373,"forks":2584,"topics":["audio-generation","cantonese","chatbot","chatgpt","chinese","cosyvoice","cross-lingual","english","fine-grained","fine-tuning","gpt-4o","japanese","korean","multi-lingual","natural-language-generation","python","text-to-speech","tts","voice-cloning"],"archived":false,"github_pushed_at":"2026-05-25T18:15:40+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/funaudiollm-cosyvoice","markdown_url":"https://www.graphcanon.com/tools/funaudiollm-cosyvoice.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/funaudiollm-cosyvoice","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=funaudiollm-cosyvoice"},{"slug":"nukeop-nuclear","name":"nuclear","tagline":"Streaming music player that finds free music for you","github_url":"https://github.com/nukeop/nuclear","owner":"nukeop","repo":"nuclear","owner_avatar_url":"https://avatars.githubusercontent.com/u/12746779?v=4","primary_language":"TypeScript","stars":18297,"forks":1321,"topics":["agent","ai","desktop-app","linux","mac","mcp","mcp-server","music","music-player","react","rust","spotify","streaming","tauri","typescript","windows"],"archived":false,"github_pushed_at":"2026-08-16T00:50:31+00:00","maintenance_label":"Very active","stars_delta_30d":235,"url":"https://www.graphcanon.com/tools/nukeop-nuclear","markdown_url":"https://www.graphcanon.com/tools/nukeop-nuclear.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nukeop-nuclear","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nukeop-nuclear"},{"slug":"cactus-compute-cactus","name":"cactus","tagline":"Low-latency AI engine for mobile devices & wearables","github_url":"https://github.com/cactus-compute/cactus","owner":"cactus-compute","repo":"cactus","owner_avatar_url":"https://avatars.githubusercontent.com/u/196640840?v=4","primary_language":"C++","stars":5535,"forks":450,"topics":["ai","android","arm","edge","edge-ai","framework","ios","llamacpp","llm","llm-inference","llms","mobile","mobile-inference","on-device-ai","quantiz","rag","smartphone","speech","transformer","whisper"],"archived":false,"github_pushed_at":"2026-07-24T15:43:41+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/cactus-compute-cactus","markdown_url":"https://www.graphcanon.com/tools/cactus-compute-cactus.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/cactus-compute-cactus","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=cactus-compute-cactus"},{"slug":"2noise-chattts","name":"ChatTTS","tagline":"A generative speech model for daily dialogue","github_url":"https://github.com/2noise/ChatTTS","owner":"2noise","repo":"ChatTTS","owner_avatar_url":"https://avatars.githubusercontent.com/u/164844019?v=4","primary_language":"Python","stars":39768,"forks":4257,"topics":["agent","chat","chatgpt","chattts","chinese","chinese-language","english","english-language","gpt","llm","llm-agent","natural-language-inference","python","text-to-speech","torch","torchaudio","tts"],"archived":false,"github_pushed_at":"2026-04-10T16:33:48+00:00","maintenance_label":"Slowing","stars_delta_30d":140,"url":"https://www.graphcanon.com/tools/2noise-chattts","markdown_url":"https://www.graphcanon.com/tools/2noise-chattts.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/2noise-chattts","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=2noise-chattts"},{"slug":"milvus-io-bootcamp","name":"bootcamp","tagline":"Dealing with all unstructured data including reverse image search, audio search, molecular search, video analysis, and question-answer systems.","github_url":"https://github.com/milvus-io/bootcamp","owner":"milvus-io","repo":"bootcamp","owner_avatar_url":"https://avatars.githubusercontent.com/u/51735404?v=4","primary_language":"Jupyter Notebook","stars":2443,"forks":684,"topics":["audio-search","deep-learning","embeddings","image-classification","image-recognition","image-search","llm","milvus","nlp","python","question-answering","rag","semantic-search","unstructured-data","vector-database"],"archived":false,"github_pushed_at":"2026-08-11T02:10:46+00:00","maintenance_label":"Active","stars_delta_30d":4,"url":"https://www.graphcanon.com/tools/milvus-io-bootcamp","markdown_url":"https://www.graphcanon.com/tools/milvus-io-bootcamp.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/milvus-io-bootcamp","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=milvus-io-bootcamp"},{"slug":"speechbrain-speechbrain","name":"speechbrain","tagline":"A PyTorch-based Speech Toolkit","github_url":"https://github.com/speechbrain/speechbrain","owner":"speechbrain","repo":"speechbrain","owner_avatar_url":"https://avatars.githubusercontent.com/u/54749030?v=4","primary_language":"Python","stars":11725,"forks":1712,"topics":["asr","audio","audio-processing","deep-learning","huggingface","language-model","pytorch","speaker-diarization","speaker-recognition","speaker-verification","speech-enhancement","speech-processing","speech-recognition","speech-separation","speech-to-text","speech-toolkit","speechrecognition","spoken-language-understanding","transformers","voice-recognition"],"archived":false,"github_pushed_at":"2026-06-15T11:24:25+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/speechbrain-speechbrain","markdown_url":"https://www.graphcanon.com/tools/speechbrain-speechbrain.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/speechbrain-speechbrain","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=speechbrain-speechbrain"},{"slug":"openai-whisper","name":"whisper","tagline":"Robust Speech Recognition via Large-Scale Weak Supervision","github_url":"https://github.com/openai/whisper","owner":"openai","repo":"whisper","owner_avatar_url":"https://avatars.githubusercontent.com/u/14957082?v=4","primary_language":"Python","stars":106740,"forks":12971,"topics":[],"archived":false,"github_pushed_at":"2026-07-28T20:18:29+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/openai-whisper","markdown_url":"https://www.graphcanon.com/tools/openai-whisper.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openai-whisper","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openai-whisper"},{"slug":"dograh-hq-dograh","name":"dograh","tagline":"Self-hosted open source voice AI platform","github_url":"https://github.com/dograh-hq/dograh","owner":"dograh-hq","repo":"dograh","owner_avatar_url":"https://avatars.githubusercontent.com/u/191861025?v=4","primary_language":"Python","stars":5064,"forks":1185,"topics":["ai-calling","asterisk-ari","conversational-ai","inbound-calls","local-llm","no-code","on-prem-voice-agent-platform","open-source","open-source-voice-ai","outbound-calls","pipecat","python","self-hosted","speech-to-speech","speech-to-text","telephony","text-to-speech","vapi-alternative","voice-agents","voice-ai-platform"],"archived":false,"github_pushed_at":"2026-07-29T10:51:41+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/dograh-hq-dograh","markdown_url":"https://www.graphcanon.com/tools/dograh-hq-dograh.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/dograh-hq-dograh","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=dograh-hq-dograh"},{"slug":"osmantic-ods","name":"ODS","tagline":"Transform personal computers into AI servers.","github_url":"https://github.com/Osmantic/ODS","owner":"Osmantic","repo":"ODS","owner_avatar_url":"https://avatars.githubusercontent.com/u/262014141?v=4","primary_language":"Python","stars":3799,"forks":551,"topics":["ai-agents","amd","comfyui","docker","llama-cpp","llm","local-ai","n8n","nvidia","open-webui","rag","self-hosted","speech-to-text","strix-halo","text-to-speech","workflow-automation"],"archived":false,"github_pushed_at":"2026-07-29T11:57:22+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/osmantic-ods","markdown_url":"https://www.graphcanon.com/tools/osmantic-ods.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/osmantic-ods","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=osmantic-ods"},{"slug":"szczyglis-dev-py-gpt","name":"py-gpt","tagline":"Desktop AI Assistant powered by multiple LLMs and various functionalities","github_url":"https://github.com/szczyglis-dev/py-gpt","owner":"szczyglis-dev","repo":"py-gpt","owner_avatar_url":"https://avatars.githubusercontent.com/u/61396542?v=4","primary_language":"Python","stars":1880,"forks":334,"topics":["ai","ai-assistant","artificial-intelligence","autonomous-agent","chatbot","claude","deepseek","desktop-app","gemini","gpt-4","gpt-5","grok","llama-index","llm","mcp","o1","ollama","openai","perplexity","sora2"],"archived":false,"github_pushed_at":"2026-08-13T19:53:38+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/szczyglis-dev-py-gpt","markdown_url":"https://www.graphcanon.com/tools/szczyglis-dev-py-gpt.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/szczyglis-dev-py-gpt","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=szczyglis-dev-py-gpt"},{"slug":"zackriya-solutions-meetily","name":"meetily","tagline":"Privacy first, AI meeting assistant with local processing","github_url":"https://github.com/Zackriya-Solutions/meetily","owner":"Zackriya-Solutions","repo":"meetily","owner_avatar_url":"https://avatars.githubusercontent.com/u/82556810?v=4","primary_language":"Rust","stars":29268,"forks":3115,"topics":["ai","ai-meeting-assistant","llm","local-ai","mac","meeting-minutes","meeting-notes","offline-first","ollama","parakeet","privacy-focused","privacy-tools","rust","self-hosted","sortformer","speech-to-text","transcription","whisper","whisper-cpp","windows"],"archived":false,"github_pushed_at":"2026-06-05T13:53:17+00:00","maintenance_label":"Steady","stars_delta_30d":4047,"url":"https://www.graphcanon.com/tools/zackriya-solutions-meetily","markdown_url":"https://www.graphcanon.com/tools/zackriya-solutions-meetily.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/zackriya-solutions-meetily","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=zackriya-solutions-meetily"},{"slug":"blaizzy-mlx-audio","name":"mlx-audio","tagline":"A text-to-speech (TTS), speech-to-text (STT) and speech-to-speech (STS) library on Apple's MLX framework.","github_url":"https://github.com/Blaizzy/mlx-audio","owner":"Blaizzy","repo":"mlx-audio","owner_avatar_url":"https://avatars.githubusercontent.com/u/23445657?v=4","primary_language":"Python","stars":7639,"forks":680,"topics":["apple-silicon","audio-processing","mlx","multimodal","speech-recognition","speech-synthesis","speech-to-text","text-to-speech","transformers"],"archived":false,"github_pushed_at":"2026-07-28T16:14:36+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/blaizzy-mlx-audio","markdown_url":"https://www.graphcanon.com/tools/blaizzy-mlx-audio.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/blaizzy-mlx-audio","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=blaizzy-mlx-audio"},{"slug":"remsky-kokoro-fastapi","name":"Kokoro-FastAPI","tagline":"Dockerized FastAPI wrapper for Kokoro-82M text-to-speech model","github_url":"https://github.com/remsky/Kokoro-FastAPI","owner":"remsky","repo":"Kokoro-FastAPI","owner_avatar_url":"https://avatars.githubusercontent.com/u/25017870?v=4","primary_language":"Python","stars":5265,"forks":858,"topics":["fastapi","huggingface-spaces","kokoro","kokoro-tts","openai-compatible-api","openwebui","pytorch","sillytavern","text-to-speech","tts","tts-api","uv"],"archived":false,"github_pushed_at":"2026-07-21T03:16:48+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/remsky-kokoro-fastapi","markdown_url":"https://www.graphcanon.com/tools/remsky-kokoro-fastapi.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/remsky-kokoro-fastapi","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=remsky-kokoro-fastapi"},{"slug":"nvidia-nemo-speech","name":"Speech","tagline":"A scalable generative AI framework for Speech AI","github_url":"https://github.com/NVIDIA-NeMo/Speech","owner":"NVIDIA-NeMo","repo":"Speech","owner_avatar_url":"https://avatars.githubusercontent.com/u/213689629?v=4","primary_language":"Python","stars":17940,"forks":3533,"topics":["asr","deeplearning","generative-ai","machine-translation","neural-networks","speaker-diariazation","speaker-recognition","speech-synthesis","speech-translation","tts"],"archived":false,"github_pushed_at":"2026-08-07T05:34:13+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/nvidia-nemo-speech","markdown_url":"https://www.graphcanon.com/tools/nvidia-nemo-speech.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/nvidia-nemo-speech","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=nvidia-nemo-speech"},{"slug":"argmaxinc-argmax-oss-swift","name":"argmax-oss-swift","tagline":"On-device Speech AI for Apple Silicon","github_url":"https://github.com/argmaxinc/argmax-oss-swift","owner":"argmaxinc","repo":"argmax-oss-swift","owner_avatar_url":"https://avatars.githubusercontent.com/u/150409474?v=4","primary_language":"Swift","stars":6294,"forks":591,"topics":["inference","ios","macos","pyannote","qwen3-tts","speaker-diarization","speakerkit","speech-recognition","speech-to-text","swift","text-to-speech","transformers","ttskit","whisper","whisperkit"],"archived":false,"github_pushed_at":"2026-07-28T23:05:28+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/argmaxinc-argmax-oss-swift","markdown_url":"https://www.graphcanon.com/tools/argmaxinc-argmax-oss-swift.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/argmaxinc-argmax-oss-swift","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=argmaxinc-argmax-oss-swift"},{"slug":"openwhispr-openwhispr","name":"openwhispr","tagline":"Voice-to-text dictation app with local and cloud models","github_url":"https://github.com/OpenWhispr/openwhispr","owner":"OpenWhispr","repo":"openwhispr","owner_avatar_url":"https://avatars.githubusercontent.com/u/254803777?v=4","primary_language":"JavaScript","stars":5018,"forks":715,"topics":["ai","anthropic","cross-platform","gemini","groq","linux","macos","nvidia","open-source","openai","parakeet","speech-to-text","transcribe","whisper","windows"],"archived":false,"github_pushed_at":"2026-07-30T07:57:27+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/openwhispr-openwhispr","markdown_url":"https://www.graphcanon.com/tools/openwhispr-openwhispr.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openwhispr-openwhispr","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openwhispr-openwhispr"},{"slug":"collabora-whisperlive","name":"WhisperLive","tagline":"A nearly-live implementation of OpenAI's Whisper for real-time voice recognition","github_url":"https://github.com/collabora/WhisperLive","owner":"collabora","repo":"WhisperLive","owner_avatar_url":"https://avatars.githubusercontent.com/u/4499761?v=4","primary_language":"Python","stars":4190,"forks":574,"topics":["dictation","obs","openai","openvino","openvino-intel","rocm","tensorrt","tensorrt-llm","text-to-speech","translation","voice-recognition","whisper","whisper-tensorrt"],"archived":false,"github_pushed_at":"2026-07-27T20:39:23+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/collabora-whisperlive","markdown_url":"https://www.graphcanon.com/tools/collabora-whisperlive.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/collabora-whisperlive","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=collabora-whisperlive"},{"slug":"buxuku-smartsub","name":"SmartSub","tagline":"A desktop subtitle tool supporting ASR and translation","github_url":"https://github.com/buxuku/SmartSub","owner":"buxuku","repo":"SmartSub","owner_avatar_url":"https://avatars.githubusercontent.com/u/7866330?v=4","primary_language":"TypeScript","stars":4624,"forks":335,"topics":["deepseek","faster-whisper","fireredasr","funasr","ollama","openai","qwen3-asr","sherpa-onnx","subtitle","translate","whisper","whisper-cpp"],"archived":false,"github_pushed_at":"2026-08-10T03:17:09+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/buxuku-smartsub","markdown_url":"https://www.graphcanon.com/tools/buxuku-smartsub.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/buxuku-smartsub","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=buxuku-smartsub"},{"slug":"funaudiollm-sensevoice","name":"SenseVoice","tagline":"Multilingual speech understanding toolkit with ASR, emotion recognition, and audio event detection.","github_url":"https://github.com/FunAudioLLM/SenseVoice","owner":"FunAudioLLM","repo":"SenseVoice","owner_avatar_url":"https://avatars.githubusercontent.com/u/167062371?v=4","primary_language":"C","stars":8961,"forks":799,"topics":["asr","audio-analysis","audio-event-detection","cantonese","cross-lingual","emotion-detection","funasr","language-identification","llama-cpp","multilingual","multilingual-asr","pytorch","sensevoice","speech-emotion-recognition","speech-recognition","speech-to-text","speech-understanding","transcription","voice-ai","whisper-alternative"],"archived":false,"github_pushed_at":"2026-07-27T14:04:29+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/funaudiollm-sensevoice","markdown_url":"https://www.graphcanon.com/tools/funaudiollm-sensevoice.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/funaudiollm-sensevoice","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=funaudiollm-sensevoice"},{"slug":"openmoss-moss-tts","name":"MOSS-TTS","tagline":"An open-source speech and sound generation model family designed for high-fidelity scenarios including multi-speaker dialogue。","github_url":"https://github.com/OpenMOSS/MOSS-TTS","owner":"OpenMOSS","repo":"MOSS-TTS","owner_avatar_url":"https://avatars.githubusercontent.com/u/156300419?v=4","primary_language":"Python","stars":3922,"forks":350,"topics":["audio","audio-tokenizer","llm","multimodal","text-to-speech","voice-cloning"],"archived":false,"github_pushed_at":"2026-07-26T11:27:15+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/openmoss-moss-tts","markdown_url":"https://www.graphcanon.com/tools/openmoss-moss-tts.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openmoss-moss-tts","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openmoss-moss-tts"},{"slug":"moonintheriver-diffsinger","name":"DiffSinger","tagline":"Singing Voice Synthesis via Shallow Diffusion Mechanism","github_url":"https://github.com/MoonInTheRiver/DiffSinger","owner":"MoonInTheRiver","repo":"DiffSinger","owner_avatar_url":"https://avatars.githubusercontent.com/u/32165188?v=4","primary_language":"Python","stars":4834,"forks":826,"topics":["aaai2022","diffusion-model","diffusion-speedup","midi","singing-synthesis","singing-voice","singing-voice-database","singing-voice-synthesis","speech-synthesis","text-to-speech","tts"],"archived":false,"github_pushed_at":"2026-07-24T07:22:44+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/moonintheriver-diffsinger","markdown_url":"https://www.graphcanon.com/tools/moonintheriver-diffsinger.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/moonintheriver-diffsinger","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=moonintheriver-diffsinger"},{"slug":"espnet-espnet","name":"espnet","tagline":"End-to-End Speech Processing Toolkit","github_url":"https://github.com/espnet/espnet","owner":"espnet","repo":"espnet","owner_avatar_url":"https://avatars.githubusercontent.com/u/34493687?v=4","primary_language":"Python","stars":9903,"forks":2421,"topics":["chainer","deep-learning","end-to-end","kaldi","machine-translation","pytorch","singing-voice-synthesis","speaker-diarization","speech-enhancement","speech-recognition","speech-separation","speech-synthesis","speech-translation","spoken-language-understanding","text-to-speech","voice-conversion"],"archived":false,"github_pushed_at":"2026-07-28T14:36:55+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/espnet-espnet","markdown_url":"https://www.graphcanon.com/tools/espnet-espnet.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/espnet-espnet","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=espnet-espnet"},{"slug":"openbmb-voxcpm","name":"VoxCPM","tagline":"Tokenizer-Free TTS for Multilingual Speech Generation, Creative Voice Design, and True-to-Life Cloning","github_url":"https://github.com/OpenBMB/VoxCPM","owner":"OpenBMB","repo":"VoxCPM","owner_avatar_url":"https://avatars.githubusercontent.com/u/89920203?v=4","primary_language":"Python","stars":34452,"forks":3939,"topics":["audio","deeplearning","minicpm","multilingual","python","pytorch","speech","speech-synthesis","text-to-speech","tts","tts-model","voice-cloning","voice-design","voxcpm"],"archived":false,"github_pushed_at":"2026-07-08T09:46:11+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/openbmb-voxcpm","markdown_url":"https://www.graphcanon.com/tools/openbmb-voxcpm.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/openbmb-voxcpm","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=openbmb-voxcpm"},{"slug":"cs-lazy-tools-chatgpt-on-cs","name":"ChatGPT-On-CS","tagline":"智能对话客服工具，支持多平台接入和多种AI模型","github_url":"https://github.com/cs-lazy-tools/ChatGPT-On-CS","owner":"cs-lazy-tools","repo":"ChatGPT-On-CS","owner_avatar_url":"https://avatars.githubusercontent.com/u/169274333?v=4","primary_language":"TypeScript","stars":4264,"forks":543,"topics":["ai","autohotkey","automation","bilibili","bot","chatgpt","chatgpt4","customer","dify","douyin","fastai","llm","pinduoduo","qianniu","wechat","wechat-bot","weibo","xiaohongshu","zhihu"],"archived":false,"github_pushed_at":"2026-06-06T03:52:08+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/cs-lazy-tools-chatgpt-on-cs","markdown_url":"https://www.graphcanon.com/tools/cs-lazy-tools-chatgpt-on-cs.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/cs-lazy-tools-chatgpt-on-cs","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=cs-lazy-tools-chatgpt-on-cs"},{"slug":"ailia-ai-ailia-models","name":"ailia-models","tagline":"Repository of pre-trained AI models for ailia SDK","github_url":"https://github.com/ailia-ai/ailia-models","owner":"ailia-ai","repo":"ailia-models","owner_avatar_url":"https://avatars.githubusercontent.com/u/55011924?v=4","primary_language":"Python","stars":2357,"forks":361,"topics":["action-recognition","anomaly-detection","audio-processing","background-removal","crowd-counting","deep-learning","embeddings","face-detection","face-recognition","fashion-ai","gan","hand-detection","image-classification","image-segmentation","llm","neural-network","object-detection","object-recognition","object-tracking","pose-estimation"],"archived":false,"github_pushed_at":"2026-07-21T14:29:01+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/ailia-ai-ailia-models","markdown_url":"https://www.graphcanon.com/tools/ailia-ai-ailia-models.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ailia-ai-ailia-models","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ailia-ai-ailia-models"},{"slug":"rvc-boss-gpt-sovits","name":"GPT-SoVITS","tagline":"Voice Cloning and Text-to-Speech with Minimal Voice Data","github_url":"https://github.com/RVC-Boss/GPT-SoVITS","owner":"RVC-Boss","repo":"GPT-SoVITS","owner_avatar_url":"https://avatars.githubusercontent.com/u/129054828?v=4","primary_language":"Python","stars":60178,"forks":6553,"topics":["text-to-speech","tts","vits","voice-clone","voice-cloneai","voice-cloning"],"archived":false,"github_pushed_at":"2026-07-22T08:21:07+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/rvc-boss-gpt-sovits","markdown_url":"https://www.graphcanon.com/tools/rvc-boss-gpt-sovits.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rvc-boss-gpt-sovits","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rvc-boss-gpt-sovits"},{"slug":"cjpais-handy","name":"Handy","tagline":"Free, open source speech-to-text application for offline use","github_url":"https://github.com/cjpais/Handy","owner":"cjpais","repo":"Handy","owner_avatar_url":"https://avatars.githubusercontent.com/u/1559480?v=4","primary_language":"Rust","stars":27869,"forks":2426,"topics":["accessibility","cross-platform","speech-to-text","tauri-v2"],"archived":false,"github_pushed_at":"2026-07-28T10:51:53+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/cjpais-handy","markdown_url":"https://www.graphcanon.com/tools/cjpais-handy.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/cjpais-handy","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=cjpais-handy"},{"slug":"open-less-openless","name":"openless","tagline":"AI-polished text from voice input for macOS and Windows","github_url":"https://github.com/Open-Less/openless","owner":"Open-Less","repo":"openless","owner_avatar_url":"https://avatars.githubusercontent.com/u/283287307?v=4","primary_language":"Rust","stars":2881,"forks":252,"topics":["ai-prompt","asr","dictation","linux","llm","macos","open-source","prompt-engineering","rust","speech-to-text","tauri","typeless","typeless-alternative","voice-input","windows","wispr-flow-alternative"],"archived":false,"github_pushed_at":"2026-07-22T00:52:10+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/open-less-openless","markdown_url":"https://www.graphcanon.com/tools/open-less-openless.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/open-less-openless","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=open-less-openless"},{"slug":"fikrikarim-parlor","name":"parlor","tagline":"On-device real-time multimodal AI for voice and vision","github_url":"https://github.com/fikrikarim/parlor","owner":"fikrikarim","repo":"parlor","owner_avatar_url":"https://avatars.githubusercontent.com/u/13066728?v=4","primary_language":"HTML","stars":1914,"forks":245,"topics":["apple-silicon","gemma","kokoro","litert-lm","local-llm","mlx","multimodal","on-device-ai","python","real-time","speech-recognition","text-to-speech","voice-assistant"],"archived":false,"github_pushed_at":"2026-07-29T15:52:20+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/fikrikarim-parlor","markdown_url":"https://www.graphcanon.com/tools/fikrikarim-parlor.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fikrikarim-parlor","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fikrikarim-parlor"},{"slug":"modelscope-funclip","name":"FunClip","tagline":"A video transcription and subtitle generation tool with LLM-assisted functionality.","github_url":"https://github.com/modelscope/FunClip","owner":"modelscope","repo":"FunClip","owner_avatar_url":"https://avatars.githubusercontent.com/u/109945100?v=4","primary_language":"Python","stars":6085,"forks":731,"topics":["ai-tools","ai-video-editing","asr","auto-subtitles","chinese","content-creation","funasr","funclip","gradio","llm","paraformer","speech-recognition","speech-to-text","subtitles-generator","transcription","video-editing","video-processing","video-subtitles","video-transcription","whisper-alternative"],"archived":false,"github_pushed_at":"2026-07-29T13:06:47+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/modelscope-funclip","markdown_url":"https://www.graphcanon.com/tools/modelscope-funclip.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/modelscope-funclip","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=modelscope-funclip"},{"slug":"huggingface-speech-to-speech","name":"speech-to-speech","tagline":"Build local voice agents with open-source models","github_url":"https://github.com/huggingface/speech-to-speech","owner":"huggingface","repo":"speech-to-speech","owner_avatar_url":"https://avatars.githubusercontent.com/u/25720743?v=4","primary_language":"Python","stars":8219,"forks":1025,"topics":["ai","assistant","language-model","machine-learning","python","speech","speech-synthesis","speech-to-text","speech-translation"],"archived":false,"github_pushed_at":"2026-07-30T11:35:50+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/huggingface-speech-to-speech","markdown_url":"https://www.graphcanon.com/tools/huggingface-speech-to-speech.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/huggingface-speech-to-speech","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=huggingface-speech-to-speech"},{"slug":"jianchang512-pyvideotrans","name":"pyvideotrans","tagline":"Translate video language and embed dubbing & subtitles","github_url":"https://github.com/jianchang512/pyvideotrans","owner":"jianchang512","repo":"pyvideotrans","owner_avatar_url":"https://avatars.githubusercontent.com/u/3378335?v=4","primary_language":"Python","stars":18478,"forks":2281,"topics":["speech-to-text","text-to-speech","video-transition"],"archived":false,"github_pushed_at":"2026-07-24T05:45:00+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/jianchang512-pyvideotrans","markdown_url":"https://www.graphcanon.com/tools/jianchang512-pyvideotrans.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/jianchang512-pyvideotrans","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=jianchang512-pyvideotrans"},{"slug":"alphacep-vosk-api","name":"vosk-api","tagline":"Offline speech recognition API","github_url":"https://github.com/alphacep/vosk-api","owner":"alphacep","repo":"vosk-api","owner_avatar_url":"https://avatars.githubusercontent.com/u/26358566?v=4","primary_language":"Jupyter Notebook","stars":15013,"forks":1740,"topics":["android","asr","deep-learning","deep-neural-networks","deepspeech","google-speech-to-text","ios","kaldi","offline","privacy","python","raspberry-pi","speaker-identification","speaker-verification","speech-recognition","speech-to-text","speech-to-text-android","stt","voice-recognition","vosk"],"archived":false,"github_pushed_at":"2026-07-02T20:03:34+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/alphacep-vosk-api","markdown_url":"https://www.graphcanon.com/tools/alphacep-vosk-api.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/alphacep-vosk-api","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=alphacep-vosk-api"},{"slug":"denizsafak-abogen","name":"abogen","tagline":"Generate audiobooks from EPUBs, PDFs and text with synchronized captions.","github_url":"https://github.com/denizsafak/abogen","owner":"denizsafak","repo":"abogen","owner_avatar_url":"https://avatars.githubusercontent.com/u/39929354?v=4","primary_language":"Python","stars":5429,"forks":398,"topics":["audiobook","audiobooks","content-creation","content-creator","ebook","epub","epub-converter","kokoro","kokoro-82m","kokoro-tts","llm","media-generation","narrator","speech-synthesis","subtitles","text-to-audio","text-to-speech","tts","voice-conversion","voice-synthesis"],"archived":false,"github_pushed_at":"2026-07-23T13:27:54+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/denizsafak-abogen","markdown_url":"https://www.graphcanon.com/tools/denizsafak-abogen.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/denizsafak-abogen","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=denizsafak-abogen"},{"slug":"systran-faster-whisper","name":"faster-whisper","tagline":"Faster Whisper transcription with CTranslate2","github_url":"https://github.com/SYSTRAN/faster-whisper","owner":"SYSTRAN","repo":"faster-whisper","owner_avatar_url":"https://avatars.githubusercontent.com/u/1520500?v=4","primary_language":"Python","stars":24689,"forks":2006,"topics":["deep-learning","inference","openai","quantization","speech-recognition","speech-to-text","transformer","whisper"],"archived":false,"github_pushed_at":"2025-11-19T14:40:46+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/systran-faster-whisper","markdown_url":"https://www.graphcanon.com/tools/systran-faster-whisper.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/systran-faster-whisper","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=systran-faster-whisper"},{"slug":"k2-fsa-sherpa-onnx","name":"sherpa-onnx","tagline":"Speech-to-text and related audio processing tools using ONNX with cross-platform support","github_url":"https://github.com/k2-fsa/sherpa-onnx","owner":"k2-fsa","repo":"sherpa-onnx","owner_avatar_url":"https://avatars.githubusercontent.com/u/71431748?v=4","primary_language":"C++","stars":13842,"forks":1596,"topics":["aarch64","android","arm32","asr","cpp","csharp","dotnet","ios","lazarus","linux","macos","mfc","object-pascal","onnx","raspberry-pi","risc-v","speech-to-text","text-to-speech","vits","windows"],"archived":false,"github_pushed_at":"2026-07-29T03:35:02+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/k2-fsa-sherpa-onnx","markdown_url":"https://www.graphcanon.com/tools/k2-fsa-sherpa-onnx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/k2-fsa-sherpa-onnx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=k2-fsa-sherpa-onnx"},{"slug":"fluidinference-fluidaudio","name":"FluidAudio","tagline":"CoreML audio models for text-to-speech, speech-to-text, voice activity detection and speaker diarization in Swift.","github_url":"https://github.com/FluidInference/FluidAudio","owner":"FluidInference","repo":"FluidAudio","owner_avatar_url":"https://avatars.githubusercontent.com/u/216957621?v=4","primary_language":"Swift","stars":2554,"forks":360,"topics":["ane","asr","audio","automatic-speech-recognition","avfoundation","coreml","ios","macos","nvidia","parakeet","real-time","speaker-diarization","speaker-embedding","speaker-identification","speaker-recognition","speech-to-text","swift","vad","voice-activity-detection"],"archived":false,"github_pushed_at":"2026-07-26T00:49:55+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/fluidinference-fluidaudio","markdown_url":"https://www.graphcanon.com/tools/fluidinference-fluidaudio.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/fluidinference-fluidaudio","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=fluidinference-fluidaudio"},{"slug":"supertone-inc-supertonic","name":"supertonic","tagline":"Lightning-Fast On-Device Multilingual TTS via ONNX","github_url":"https://github.com/supertone-inc/supertonic","owner":"supertone-inc","repo":"supertonic","owner_avatar_url":"https://avatars.githubusercontent.com/u/61007505?v=4","primary_language":"Swift","stars":13543,"forks":1427,"topics":["cpp","csharp","flutter","go","ios","java","lightweight","multilingual","nodejs","on-device","onnx","onnxruntime","python","rust","speech-synthesis","swift","text-to-speech","tts","web","webgpu"],"archived":false,"github_pushed_at":"2026-07-24T04:00:17+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/supertone-inc-supertonic","markdown_url":"https://www.graphcanon.com/tools/supertone-inc-supertonic.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/supertone-inc-supertonic","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=supertone-inc-supertonic"},{"slug":"iahispano-applio","name":"Applio","tagline":"A simple high-quality voice conversion tool focused on ease of use and performance","github_url":"https://github.com/IAHispano/Applio","owner":"IAHispano","repo":"Applio","owner_avatar_url":"https://avatars.githubusercontent.com/u/133545040?v=4","primary_language":"Python","stars":3530,"forks":563,"topics":["ai","applio","pytorch","rvc","speech","speech-to-speech","text-to-speech","tts","vc","vits","voice","voice-clone","voice-cloning","voice-conversion"],"archived":false,"github_pushed_at":"2026-07-27T11:45:48+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/iahispano-applio","markdown_url":"https://www.graphcanon.com/tools/iahispano-applio.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/iahispano-applio","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=iahispano-applio"},{"slug":"umlx5h-llplayer","name":"LLPlayer","tagline":"Media player for language learning featuring dual subtitles, AI-generated subtitles and real-time translation","github_url":"https://github.com/umlx5h/LLPlayer","owner":"umlx5h","repo":"LLPlayer","owner_avatar_url":"https://avatars.githubusercontent.com/u/20206121?v=4","primary_language":"C#","stars":4027,"forks":233,"topics":["asr","csharp","flyleaf","language-learning","llm","media-player","ocr","ollama","player","video","video-player","whisper","wpf","yt-dlp"],"archived":false,"github_pushed_at":"2026-07-19T12:20:10+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/umlx5h-llplayer","markdown_url":"https://www.graphcanon.com/tools/umlx5h-llplayer.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/umlx5h-llplayer","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=umlx5h-llplayer"},{"slug":"open-llm-vtuber-open-llm-vtuber","name":"Open-LLM-VTuber","tagline":"Voice interaction with LLMs and Live2D visuals","github_url":"https://github.com/Open-LLM-VTuber/Open-LLM-VTuber","owner":"Open-LLM-VTuber","repo":"Open-LLM-VTuber","owner_avatar_url":"https://avatars.githubusercontent.com/u/193920693?v=4","primary_language":"Python","stars":13224,"forks":1563,"topics":["ai","ai-companion","ai-vtuber","ai-waifu","chatbots","live2d","live2d-web","llm","neuro-sama","ollama"],"archived":false,"github_pushed_at":"2026-05-15T07:18:04+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/open-llm-vtuber-open-llm-vtuber","markdown_url":"https://www.graphcanon.com/tools/open-llm-vtuber-open-llm-vtuber.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/open-llm-vtuber-open-llm-vtuber","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=open-llm-vtuber-open-llm-vtuber"},{"slug":"iamsrikanthnani-pluely","name":"pluely","tagline":"Privacy-first AI assistant for meetings and interviews","github_url":"https://github.com/iamsrikanthnani/pluely","owner":"iamsrikanthnani","repo":"pluely","owner_avatar_url":"https://avatars.githubusercontent.com/u/55689131?v=4","primary_language":"TypeScript","stars":2354,"forks":491,"topics":["ai-assistant","claude","cluely-alternative","desktop-app","gemini","grok","grok-4","llm","openai","react","rust","shadcn","speech-to-text","stealth","tailwindcss","tauri","typescript","undetectable","voice-activity-detection"],"archived":false,"github_pushed_at":"2026-07-14T17:47:19+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/iamsrikanthnani-pluely","markdown_url":"https://www.graphcanon.com/tools/iamsrikanthnani-pluely.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/iamsrikanthnani-pluely","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=iamsrikanthnani-pluely"},{"slug":"rsxdalv-tts-webui","name":"TTS-WebUI","tagline":"A comprehensive WebUI for multiple TTS systems","github_url":"https://github.com/rsxdalv/TTS-WebUI","owner":"rsxdalv","repo":"TTS-WebUI","owner_avatar_url":"https://avatars.githubusercontent.com/u/6757283?v=4","primary_language":"TypeScript","stars":3216,"forks":326,"topics":["ace-step","ai","audio-generation","cosyvoice","generative-ai","generator","gradio","music","musicgen","openai-api","openvoice","rvc","styletts2","text-to-speech","tortoise-tts","tts","vocos"],"archived":false,"github_pushed_at":"2026-07-27T13:23:40+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/rsxdalv-tts-webui","markdown_url":"https://www.graphcanon.com/tools/rsxdalv-tts-webui.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/rsxdalv-tts-webui","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=rsxdalv-tts-webui"},{"slug":"babysor-mockingbird","name":"MockingBird","tagline":"Clone a voice in 5 seconds to generate arbitrary speech in real-time","github_url":"https://github.com/babysor/MockingBird","owner":"babysor","repo":"MockingBird","owner_avatar_url":"https://avatars.githubusercontent.com/u/7423248?v=4","primary_language":"Python","stars":36913,"forks":5197,"topics":["ai","deep-learning","pytorch","speech","text-to-speech","tts"],"archived":false,"github_pushed_at":"2026-03-03T14:59:58+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/babysor-mockingbird","markdown_url":"https://www.graphcanon.com/tools/babysor-mockingbird.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/babysor-mockingbird","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=babysor-mockingbird"},{"slug":"debpalash-omnivoice-studio","name":"OmniVoice-Studio","tagline":"The open-source ElevenLabs alternative for local voice cloning and related tasks","github_url":"https://github.com/debpalash/OmniVoice-Studio","owner":"debpalash","repo":"OmniVoice-Studio","owner_avatar_url":"https://avatars.githubusercontent.com/u/4178343?v=4","primary_language":"Python","stars":9209,"forks":1494,"topics":["asr","audiobook","dubbing","dubbing-ai","elevenlabs","local-ai","omnivoice","omnivoice-studio","self-hosted","speech-recognition","speech-to-text","text-to-speech","transcription","tts","video-editing","voice-ai","voice-cloning","voice-generation"],"archived":false,"github_pushed_at":"2026-07-28T19:51:31+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/debpalash-omnivoice-studio","markdown_url":"https://www.graphcanon.com/tools/debpalash-omnivoice-studio.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/debpalash-omnivoice-studio","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=debpalash-omnivoice-studio"},{"slug":"microsoft-foundry-local","name":"Foundry-Local","tagline":"SDK and CLI for local AI inference with GPU acceleration, supporting speech-to-text models like Whisper.","github_url":"https://github.com/microsoft/Foundry-Local","owner":"microsoft","repo":"Foundry-Local","owner_avatar_url":"https://avatars.githubusercontent.com/u/6154722?v=4","primary_language":"C++","stars":2479,"forks":348,"topics":["ai-sdk","chat-completions","foundry-local","gpu-acceleration","local-ai","microsoft","on-device-inference","onnx-runtime","speech-to-text","whisper"],"archived":false,"github_pushed_at":"2026-07-30T08:53:19+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/microsoft-foundry-local","markdown_url":"https://www.graphcanon.com/tools/microsoft-foundry-local.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/microsoft-foundry-local","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=microsoft-foundry-local"},{"slug":"harry0703-audionotes","name":"AudioNotes","tagline":"Quickly extracts audio and video content into structured markdown notes","github_url":"https://github.com/harry0703/AudioNotes","owner":"harry0703","repo":"AudioNotes","owner_avatar_url":"https://avatars.githubusercontent.com/u/4928832?v=4","primary_language":"Python","stars":2259,"forks":330,"topics":["ai","asr","funasr","ollama","python","qwen2","whisper"],"archived":false,"github_pushed_at":"2026-08-12T06:14:24+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/harry0703-audionotes","markdown_url":"https://www.graphcanon.com/tools/harry0703-audionotes.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/harry0703-audionotes","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=harry0703-audionotes"},{"slug":"espeak-ng-espeak-ng","name":"espeak-ng","tagline":"Open source speech synthesizer supporting more than hundred languages and accents","github_url":"https://github.com/espeak-ng/espeak-ng","owner":"espeak-ng","repo":"espeak-ng","owner_avatar_url":"https://avatars.githubusercontent.com/u/16214005?v=4","primary_language":"C","stars":6688,"forks":1263,"topics":["android","espeak","espeak-ng","speech-synthesis","text-to-speech"],"archived":false,"github_pushed_at":"2026-07-27T15:59:11+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/espeak-ng-espeak-ng","markdown_url":"https://www.graphcanon.com/tools/espeak-ng-espeak-ng.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/espeak-ng-espeak-ng","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=espeak-ng-espeak-ng"},{"slug":"index-tts-index-tts","name":"index-tts","tagline":"A Breakthrough in Emotionally Expressive and Duration-Controlled Auto-Regressive Zero-Shot Text-to-Speech","github_url":"https://github.com/index-tts/index-tts","owner":"index-tts","repo":"index-tts","owner_avatar_url":"https://avatars.githubusercontent.com/u/196291161?v=4","primary_language":"Python","stars":22239,"forks":2709,"topics":["bigvgan","cross-lingual","indextts","text-to-speech","tts","voice-clone","zero-shot-tts"],"archived":false,"github_pushed_at":"2026-07-14T11:43:37+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/index-tts-index-tts","markdown_url":"https://www.graphcanon.com/tools/index-tts-index-tts.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/index-tts-index-tts","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=index-tts-index-tts"},{"slug":"m-bain-whisperx","name":"whisperX","tagline":"WhisperX for automatic speech recognition with word-level timestamps and diarization","github_url":"https://github.com/m-bain/whisperX","owner":"m-bain","repo":"whisperX","owner_avatar_url":"https://avatars.githubusercontent.com/u/36994049?v=4","primary_language":"Python","stars":23329,"forks":2361,"topics":["asr","speech","speech-recognition","speech-to-text","whisper"],"archived":false,"github_pushed_at":"2026-07-13T08:30:07+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/m-bain-whisperx","markdown_url":"https://www.graphcanon.com/tools/m-bain-whisperx.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/m-bain-whisperx","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=m-bain-whisperx"},{"slug":"snakers4-silero-models","name":"silero-models","tagline":"Silero Models provide simple access to pre-trained text-to-speech models","github_url":"https://github.com/snakers4/silero-models","owner":"snakers4","repo":"silero-models","owner_avatar_url":"https://avatars.githubusercontent.com/u/12515440?v=4","primary_language":"Jupyter Notebook","stars":6030,"forks":369,"topics":["armenian","azerbaijani","belarus","colab","georgian","kazakh","kyrgyz","pretrained-models","pytorch","russian","speech","speech-synthesis","speech-to-text","tajik","text-to-speech","torch-hub","tts","tts-models","ukrainian","uzbek"],"archived":false,"github_pushed_at":"2026-06-04T05:33:28+00:00","maintenance_label":"Steady","url":"https://www.graphcanon.com/tools/snakers4-silero-models","markdown_url":"https://www.graphcanon.com/tools/snakers4-silero-models.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/snakers4-silero-models","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=snakers4-silero-models"},{"slug":"ccostan-home-assistantconfig","name":"Home-AssistantConfig","tagline":"Home Assistant configuration and documentation for smart home setup.","github_url":"https://github.com/CCOSTAN/Home-AssistantConfig","owner":"CCOSTAN","repo":"Home-AssistantConfig","owner_avatar_url":"https://avatars.githubusercontent.com/u/2160436?v=4","primary_language":"Python","stars":5256,"forks":503,"topics":["ai","alexa","automation","codex","codex-skills","hacktoberfest","home-assistant","home-assistant-config","home-automation","homeassistant","homeassistant-config","homeautomation","hue","led-controller","lovelace","smart-home","smarthome","youtube","youtube-videos"],"archived":false,"github_pushed_at":"2026-08-05T16:20:35+00:00","maintenance_label":"Very active","url":"https://www.graphcanon.com/tools/ccostan-home-assistantconfig","markdown_url":"https://www.graphcanon.com/tools/ccostan-home-assistantconfig.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/ccostan-home-assistantconfig","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=ccostan-home-assistantconfig"},{"slug":"filipecalegario-awesome-generative-ai","name":"awesome-generative-ai","tagline":"A comprehensive list of generative AI resources","github_url":"https://github.com/filipecalegario/awesome-generative-ai","owner":"filipecalegario","repo":"awesome-generative-ai","owner_avatar_url":"https://avatars.githubusercontent.com/u/299057?v=4","primary_language":null,"stars":3508,"forks":832,"topics":["ai-art","awesome","awesome-list","chatgpt","dall-e","dalle2","embeddings","generative-ai","gpt-4","llm","llm-agent","midjourney","openai","prompt-engineering","semantic-search","stable-diffusion","text-to-image","txt2img"],"archived":false,"github_pushed_at":"2025-12-18T07:31:25+00:00","maintenance_label":"Slowing","url":"https://www.graphcanon.com/tools/filipecalegario-awesome-generative-ai","markdown_url":"https://www.graphcanon.com/tools/filipecalegario-awesome-generative-ai.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/filipecalegario-awesome-generative-ai","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=filipecalegario-awesome-generative-ai"},{"slug":"abus-aikorea-voice-pro","name":"voice-pro","tagline":"Gradio WebUI for TTS and voice cloning with audio processing capabilities","github_url":"https://github.com/abus-aikorea/voice-pro","owner":"abus-aikorea","repo":"voice-pro","owner_avatar_url":"https://avatars.githubusercontent.com/u/161691694?v=4","primary_language":"Python","stars":11348,"forks":1663,"topics":["audiobook","faster-whisper","gradio","karaoke","podcasts","speech-recognition","speech-synthesis","speech-to-text","subtitles","text-to-speech","transcription","translator","tts","voice-cloning","voice-conversion","webui","whisper","whisperx","yt-dlp"],"archived":false,"github_pushed_at":"2026-07-13T01:28:10+00:00","maintenance_label":"Active","url":"https://www.graphcanon.com/tools/abus-aikorea-voice-pro","markdown_url":"https://www.graphcanon.com/tools/abus-aikorea-voice-pro.md","api_url":"https://www.graphcanon.com/api/graphcanon/tools/abus-aikorea-voice-pro","graph_url":"https://www.graphcanon.com/api/graphcanon/graph?tool=abus-aikorea-voice-pro"}]}}