fastapi uvicorn loguru # Qwen3-TTS - state-of-the-art TTS with voice cloning (Apache 2.0) # 1.7B params, 10 languages, 97ms latency, 12Hz tokenizer qwen-tts>=0.0.5 # OHF-Voice fork doesn't have installable Python package yet # Stick with PyPI piper-tts but use absolute paths in config # piper-tts>=1.2.0 # 🦝 RACCOON TODO: Create our own PyPI package from OHF-Voice fork # git+https://github.com/OHF-Voice/piper1-gpl.git@v1.3.0#subdirectory=src/python_run # coqui-tts[languages] # Silero TTS - actively maintained, small efficient models # Note: Silero models are loaded via torch.hub, no package install needed # Models: ~50-100MB each, CPU-friendly, real-time capable # omegaconf # Required by Silero TTS # Chatterbox - emotion control, 23 languages (Resemble AI) # Install from git since no PyPI package exists yet # 🦝 RACCOON NOTE: Disabled due to dependency conflict with Coqui TTS # gradio 5.44.1 requires typer<1.0 and >=0.12, but spacy 3.6.x requires typer<0.10.0 # TODO: Test Chatterbox in isolated environment or wait for dependency updates # git+https://github.com/resemble-ai/chatterbox.git # langdetect pyyaml # Kokoro TTS - fast decoder-only architecture # Lightweight decoder-only TTS, 82M params, 24kHz output # kokoro>=0.9.2 soundfile # Required by Qwen3-TTS and Kokoro for audio output datasets # HuggingFace datasets for voice corpus downloads torchcodec # Audio decoding for HuggingFace datasets transformers>=4.35.0 # Hugging Face Hub for model downloads huggingface-hub[cli] # Creating an environment where deepspeed works is complex, for now it will be disabled by default. #deepspeed torch; sys_platform != "darwin" torchaudio; sys_platform != "darwin" # for MPS accelerated torch on Mac - doesn't work yet, incomplete support in torch and torchaudio torch; --index-url https://download.pytorch.org/whl/cpu; sys_platform == "darwin" torchaudio; --index-url https://download.pytorch.org/whl/cpu; sys_platform == "darwin" # ROCM (Linux only) - use requirements.amd.txt