Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
61 changes: 61 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,67 @@ tts = TTS(pretrained="khanomtan")
file = tts.tts("ภาษาไทย", speaker_idx="Linda", filename="output.wav")
```

### Real-time / Streaming TTS (FastThaiG2P)

PyThaiTTS supports low-latency, real-time streaming speech synthesis with FastThaiG2P, making it ideal for conversational voice agents and LLM streaming:

#### 1. Streaming Audio from Text

Synthesize chunk-by-chunk in real time:

```python
from pythaitts import TTS

tts = TTS(pretrained="fastthaig2p")

# Stream audio chunks as 24kHz float32 NumPy arrays
for audio_chunk in tts.stream("สวัสดีครับ ยินดีต้อนรับสู่ระบบเรียลไทม์ทีทีเอส"):
print(f"Audio chunk shape: {audio_chunk.shape}")

# Stream raw 16-bit PCM bytes (for WebSockets or PyAudio)
for pcm_bytes in tts.stream("สวัสดีครับ", return_type="bytes"):
# send over websocket or write to audio stream
pass
```

#### 2. Streaming from an LLM Token Stream

Feed tokens directly from an LLM or generator into `tts.stream()`:

```python
from pythaitts import TTS

tts = TTS()

def token_stream():
tokens = ["สวัสดี", "ครับ", " ", "นี่", "คือ", "การ", "สตรีม", "มิ่ง"]
for tok in tokens:
yield tok

for audio_chunk in tts.stream(token_stream()):
# Process or play chunk with low latency
pass
```

#### 3. Integration with the RealtimeTTS Library

You can use FastThaiG2P as an engine with [KoljaB/RealtimeTTS](https://github.com/KoljaB/RealtimeTTS):

```sh
pip install pythaitts[realtime]
```

```python
from RealtimeTTS import TextToAudioStream
from pythaitts.realtime import FastThaiG2PEngine

engine = FastThaiG2PEngine()
stream = TextToAudioStream(engine)
stream.feed("สวัสดีครับ วันนี้อากาศดีมาก")
stream.play()
```


### Text Preprocessing

PyThaiTTS includes automatic text preprocessing to improve TTS quality:
Expand Down
139 changes: 139 additions & 0 deletions demo_realtime.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,139 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""
Real-time Text-to-Speech Demo for FastThaiG2P in PyThaiTTS.

Demonstrates:
1. Low-latency streaming TTS from text string (chunk by chunk).
2. Streaming TTS from an LLM-like token generator.
3. Streaming 16-bit PCM bytes for audio pipelines/WebSockets.
4. FastThaiG2PEngine integration with KoljaB/RealtimeTTS.
"""

import time
from pythaitts import TTS, RealtimeTTS, FastThaiG2PEngine


def main():
print("=" * 65)
print("PyThaiTTS - Real-time TTS Demo (FastThaiG2P)")
print("=" * 65)
print()

# 1. Initialize TTS model
print("Initializing FastThaiG2P TTS model...")
t_start = time.time()
tts = TTS(pretrained="fastthaig2p")
print(f"✓ Model loaded in {time.time() - t_start:.2f}s")
print()

# 2. Streaming from a full text string
sample_text = (
"สวัสดีครับ ยินดีต้อนรับสู่ระบบสังเคราะห์เสียงภาษาไทยแบบเรียลไทม์ "
"ระบบนี้ช่วยให้สร้างเสียงพูดได้อย่างรวดเร็วและต่อเนื่อง "
"โดยเริ่มส่งสัญญาณเสียงได้ทันทีตั้งแต่ข้อความส่วนแรกประมวลผลเสร็จ"
)
print("-----------------------------------------------------------------")
print("Demo 1: Real-time Streaming from Full Text")
print("-----------------------------------------------------------------")
print(f"Input text:\n{sample_text}\n")

t0 = time.time()
total_audio_samples = 0
chunk_count = 0
ttfa = None # Time to First Audio

for audio_chunk in tts.stream(sample_text, return_type="waveform"):
chunk_count += 1
elapsed = time.time() - t0
if ttfa is None:
ttfa = elapsed
print(f"⚡ Time to First Audio (TTFA): {ttfa * 1000:.1f} ms!")

duration = len(audio_chunk) / 24000.0
total_audio_samples += len(audio_chunk)
print(
f" [Chunk {chunk_count}] Received {len(audio_chunk)} samples "
f"({duration:.2f}s of audio) at +{elapsed:.2f}s"
)

total_time = time.time() - t0
total_audio_sec = total_audio_samples / 24000.0
rtf = total_time / total_audio_sec if total_audio_sec > 0 else 0
print(f"\n✓ Generated {total_audio_sec:.2f}s of audio across {chunk_count} chunks in {total_time:.2f}s")
print(f" Real-Time Factor (RTF): {rtf:.3f} (< 1.0 means faster than real-time)")
print()

# 3. Streaming from simulated LLM token stream
print("-----------------------------------------------------------------")
print("Demo 2: Real-time Streaming from LLM Token Stream")
print("-----------------------------------------------------------------")

def simulate_llm_stream():
tokens = [
"สวัสดี", "ครับ", " ", "นี่", "คือ", "การ", "ทดสอบ", " ",
"การ", "สตรีม", "มิ่ง", " ", "ข้อความ", "จาก", " ", "โมเดล",
"ภาษา", "ขนาด", "ใหญ่", " ", "แบบ", "เรียล", "ไทม์", "ครับ"
]
print("Streaming tokens from LLM: ", end="", flush=True)
for tok in tokens:
print(tok, end="", flush=True)
time.sleep(0.04) # simulate LLM generation delay
yield tok
print("\n")

t0 = time.time()
stream_chunks = 0
for audio_chunk in tts.stream(simulate_llm_stream(), return_type="waveform"):
stream_chunks += 1
elapsed = time.time() - t0
dur = len(audio_chunk) / 24000.0
print(f" [Audio Chunk {stream_chunks}] Duration: {dur:.2f}s at +{elapsed:.2f}s")

print(f"✓ LLM streaming synthesis complete in {time.time() - t0:.2f}s\n")

# 4. Streaming 16-bit PCM bytes
print("-----------------------------------------------------------------")
print("Demo 3: Streaming 16-bit PCM Bytes (for WebSockets / PyAudio)")
print("-----------------------------------------------------------------")
byte_chunks = 0
total_bytes = 0
for pcm in tts.stream("ระบบเสียงภาษาไทย คุณภาพสูง", return_type="bytes"):
byte_chunks += 1
total_bytes += len(pcm)
print(f" [PCM Chunk {byte_chunks}] Received {len(pcm)} bytes of raw 16-bit PCM")

print(f"✓ Total raw PCM data: {total_bytes} bytes (24 kHz, 16-bit, mono)\n")

# 5. RealtimeTTS Library compatibility
print("-----------------------------------------------------------------")
print("Demo 4: RealtimeTTS Engine (KoljaB/RealtimeTTS compatibility)")
print("-----------------------------------------------------------------")
try:
from RealtimeTTS import TextToAudioStream
print("RealtimeTTS library detected. Initializing FastThaiG2PEngine...")
engine = FastThaiG2PEngine()
stream = TextToAudioStream(engine)
print("✓ FastThaiG2PEngine ready with RealtimeTTS TextToAudioStream!")
except ImportError:
print("KoljaB/RealtimeTTS library is not installed.")
print("To use with RealtimeTTS:")
print(" pip install pythaitts[realtime]")
print(" or: pip install RealtimeTTS")
print("\nFastThaiG2PEngine can then be used directly:")
print(" from RealtimeTTS import TextToAudioStream")
print(" from pythaitts.realtime import FastThaiG2PEngine")
print(" engine = FastThaiG2PEngine()")
print(" stream = TextToAudioStream(engine)")
print(" stream.feed('สวัสดีครับ')")
print(" stream.play()")

print()
print("=" * 65)
print("Real-time TTS Demo completed successfully!")
print("=" * 65)
return 0


if __name__ == "__main__":
exit(main())
30 changes: 30 additions & 0 deletions example_realtime.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""
Example for Real-time TTS using FastThaiG2P.

Usage:
python example_realtime.py
"""

from pythaitts import TTS

def main():
print("PyThaiTTS Realtime TTS Example (fastthaig2p)")
print("-" * 50)

# Initialize FastThaiG2P TTS
tts = TTS(pretrained="fastthaig2p")

# 1. Stream speech from text string chunk-by-chunk
text = "สวัสดีครับ ยินดีต้อนรับสู่ระบบสังเคราะห์เสียงภาษาไทยแบบเรียลไทม์"
print(f"Streaming text: {text}\n")

for i, audio_chunk in enumerate(tts.stream(text, return_type="waveform"), start=1):
duration = len(audio_chunk) / 24000.0
print(f" Chunk {i}: {len(audio_chunk)} samples ({duration:.2f}s of audio)")

print("\nDone!")

if __name__ == "__main__":
main()
77 changes: 77 additions & 0 deletions pythaitts/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,17 @@
"""
__version__ = "0.6.0"

from typing import Union, Iterable, Iterator

from pythaitts.preprocess import preprocess_text, num_to_thai, expand_maiyamok
from pythaitts.realtime import (
RealtimeTTS,
FastThaiG2PEngine,
PyThaiTTSEngine,
RealtimeTTSEngine,
chunk_text,
stream_text_to_chunks,
)


class TTS:
Expand Down Expand Up @@ -100,3 +110,70 @@ def tts(self, text: str, speaker_idx: str = "thai_som", language_idx: str = "th-
return_type=return_type,
filename=filename
)

def stream(
self,
text: Union[str, Iterable[str]],
speaker_idx: str = "thai_som",
return_type: str = "waveform",
play: bool = False,
preprocess: bool = True,
max_phonemes: int = 400,
**kwargs,
):
"""
Stream speech synthesis in real-time.

:param Union[str, Iterable[str]] text: Input text or stream of text tokens (e.g. from LLM)
:param str speaker_idx: Voice to use (default: "thai_som" for fastthaig2p)
:param str return_type: Return format ("waveform", "bytes", "raw", "file")
:param bool play: Whether to play audio chunks in real-time to speakers
:param bool preprocess: Whether to preprocess text (numbers to words, ๆ)
:param int max_phonemes: Maximum phonemes per synthesized chunk
:param kwargs: Additional parameters passed to the model
:return: Generator yielding audio chunks
"""
if self.pretrained in ("fastthaig2p", "FastThaiG2P"):
if speaker_idx in ("Linda", None):
speaker_idx = "thai_som"
return self.model.stream(
text=text,
speaker_idx=speaker_idx,
return_type=return_type,
play=play,
preprocess=preprocess,
max_phonemes=max_phonemes,
**kwargs,
)
else:
from pythaitts.realtime import stream_text_to_chunks

def _gen():
for chunk in stream_text_to_chunks(
text, max_phonemes=max_phonemes, preprocess=preprocess
):
yield self.tts(
text=chunk,
speaker_idx=speaker_idx,
return_type=return_type,
preprocess=False,
**kwargs,
)

return _gen()

tts_stream = stream


__all__ = [
"TTS",
"RealtimeTTS",
"FastThaiG2PEngine",
"PyThaiTTSEngine",
"RealtimeTTSEngine",
"preprocess_text",
"num_to_thai",
"expand_maiyamok",
"chunk_text",
"stream_text_to_chunks",
]
22 changes: 21 additions & 1 deletion pythaitts/pretrained/fastthaig2p/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,13 +23,33 @@
from .normalizer import normalize
from .tokenizer import Tokenizer

__all__ = ["G2P", "Tokenizer", "normalize", "ipa_to_kokoro", "TTS", "FastThaiG2P"]
__all__ = [
"G2P",
"Tokenizer",
"normalize",
"ipa_to_kokoro",
"TTS",
"FastThaiG2P",
"FastThaiG2PEngine",
"FastThaiG2PVoice",
"chunk_text",
"stream_text_to_chunks",
]


def __getattr__(name):
if name in ("TTS", "FastThaiG2P"):
from .tts import TTS, FastThaiG2P

return FastThaiG2P if name == "FastThaiG2P" else TTS
if name in ("FastThaiG2PEngine", "FastThaiG2PVoice", "chunk_text", "stream_text_to_chunks"):
from pythaitts.realtime import (
FastThaiG2PEngine,
FastThaiG2PVoice,
chunk_text,
stream_text_to_chunks,
)

return locals()[name]
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")

Loading
Loading