[{"data":1,"prerenderedAt":-1},["ShallowReactive",2],{"casestudy-development-tts-compare":3},{"id":4,"title":5,"body":6,"client":197,"credits":198,"description":190,"extension":207,"featured":208,"gallery":209,"geo":213,"keyart":215,"lede":217,"mediaType":218,"meta":220,"navigation":221,"path":222,"seo":223,"sort":224,"status":225,"stem":226,"tags":227,"vertical":234,"year":235,"yt_video":236,"__hash__":237},"casestudies\u002Fwork\u002Fdevelopment-tts-compare.md","TTS Compare",{"type":7,"value":8,"toc":189},"minimark",[9,14,18,22,25,131,135,173,177],[10,11,13],"h2",{"id":12},"overview","Overview",[15,16,17],"p",{},"TTS Compare lets you generate the same input across multiple open-source text-to-speech models in a single run, then compare quality, latency, and per-model feature support side-by-side. The project addresses a recurring practical question, \"which TTS should we ship?\", by replacing ad-hoc one-off testing with a reproducible harness.",[10,19,21],{"id":20},"models-covered","Models covered",[15,23,24],{},"The harness currently runs eight models from across the modern open-source TTS field:",[26,27,28,44],"table",{},[29,30,31],"thead",{},[32,33,34,38,41],"tr",{},[35,36,37],"th",{},"Model",[35,39,40],{},"Params",[35,42,43],{},"Notable feature",[45,46,47,59,70,81,91,102,112,122],"tbody",{},[32,48,49,53,56],{},[50,51,52],"td",{},"Maya1",[50,54,55],{},"3B",[50,57,58],{},"17 emotion tags, voice description",[32,60,61,64,67],{},[50,62,63],{},"Kokoro-82M",[50,65,66],{},"82M",[50,68,69],{},"11 voice presets, very small",[32,71,72,75,78],{},[50,73,74],{},"Chatterbox",[50,76,77],{},"500M",[50,79,80],{},"5 emotion tags, reference-audio cloning",[32,82,83,86,88],{},[50,84,85],{},"Orpheus-3B",[50,87,55],{},[50,89,90],{},"vLLM backend (Linux only)",[32,92,93,96,99],{},[50,94,95],{},"Qwen3-TTS",[50,97,98],{},"1.7B",[50,100,101],{},"Voice description",[32,103,104,107,109],{},[50,105,106],{},"Fish Speech 1.5",[50,108,77],{},[50,110,111],{},"Reference-audio cloning",[32,113,114,117,119],{},[50,115,116],{},"CosyVoice 2",[50,118,77],{},[50,120,121],{},"Voice description + reference audio",[32,123,124,127,129],{},[50,125,126],{},"XTTS v2",[50,128,77],{},[50,130,111],{},[10,132,134],{"id":133},"architecture-decisions","Architecture decisions",[136,137,138,155,161,167],"ul",{},[139,140,141,150,151,154],"li",{},[142,143,144,145,149],"strong",{},"Per-model isolated ",[146,147,148],"code",{},".venv","."," Each model gets its own environment under ",[146,152,153],{},"models\u002F\u003Cname>\u002F"," to avoid CUDA\u002Ftorch version skew between systems with conflicting requirements.",[139,156,157,160],{},[142,158,159],{},"JSON-over-stdin worker protocol."," Each model is invoked as a subprocess that accepts a JSON request on stdin and writes a WAV file. Decoupling the harness from the model runtimes keeps the Textual TUI snappy and lets a model crash without taking the rest of the run down.",[139,162,163,166],{},[142,164,165],{},"Hardware adaptive."," Detects CUDA, MPS, or CPU and configures models accordingly.",[139,168,169,172],{},[142,170,171],{},"Three-screen TUI."," Input → model selection → execution, with real-time logs streaming as each model generates speech.",[10,174,176],{"id":175},"references","References",[136,178,179],{},[139,180,181,182],{},"Repository: ",[183,184,188],"a",{"href":185,"rel":186},"https:\u002F\u002Fgithub.com\u002Fandrewmarconi\u002Ftts-compare",[187],"nofollow","github.com\u002Fandrewmarconi\u002Ftts-compare",{"title":190,"searchDepth":191,"depth":191,"links":192},"",2,[193,194,195,196],{"id":12,"depth":191,"text":13},{"id":20,"depth":191,"text":21},{"id":133,"depth":191,"text":134},{"id":175,"depth":191,"text":176},"Personal Project",[199,204],{"sort":200,"role":201,"value":202,"uri":203},1,"Creator & Developer","Andrew Marconi","https:\u002F\u002Flinkedin.com\u002Fin\u002Fmarconi",{"sort":205,"role":206,"value":188,"uri":185},10,"Repository","md",false,[210],{"src":211,"alt":212},"\u002Fwork\u002Fdevelopment-tts-compare\u002Ftts-matrix-models.jpg","Model comparison matrix",[214],"Global",{"src":216,"alt":5},"\u002Fwork\u002Fdevelopment-tts-compare\u002Ftts-matrix-keyart.jpg","A side-by-side comparison framework for eight open-source text-to-speech models, with an interactive Textual TUI for evaluating quality, latency, and feature coverage.",[219],"CLI Tool",{},true,"\u002Fwork\u002Fdevelopment-tts-compare",{"title":5,"description":190},200,"live","work\u002Fdevelopment-tts-compare",[228,229,230,231,232,233],"Python","TTS","Textual","Benchmarking","Generative AI","Open Source","AI\u002FMachine Learning",2026,null,"gEl6Sqdh1VEnRZqMkeubsrkqtlhbpKiQL6U2C3OkvhA"]