Source code for pipecat.utils.text.phonemes

#
# Copyright (c) 2024-2026, Daily
#
# SPDX-License-Identifier: BSD 2-Clause License
#

"""Shared pronunciation parsing and normalization utilities for TTS services.

A pronunciation reaches a TTS service as an IPA string. Services each want their
own markup for it, but they all start from the same steps: tidy the notation,
split it into phones, and find which vowel each stress mark belongs to. Those
steps live here, so a service's formatter only has to render the result.

Normalization is notation-only. It unifies ways of writing the same symbol
(``ʧ`` and ``tʃ``, ``g`` and ``ɡ``, ``'`` and ``ˈ``) and never changes which sound
is written, so it is safe for any language.
"""

import unicodedata

PRIMARY_STRESS = "ˈ"
SECONDARY_STRESS = "ˌ"
STRESS_MARKS = (PRIMARY_STRESS, SECONDARY_STRESS)

TIE_BAR = "\u0361"

# Ligatures and ASCII look-alikes of IPA symbols. "tʃ" and "dʒ" are always read
# as one phone, so they are written untied; other affricates keep the tie bar
# that tells them apart from a cluster.
_IPA_NOTATION = [
    ("\u035c", TIE_BAR),  # tie bar below
    ("ʧ", "tʃ"),
    ("ʤ", "dʒ"),
    (f"t{TIE_BAR}ʃ", "tʃ"),
    (f"d{TIE_BAR}ʒ", "dʒ"),
    ("ʦ", f"t{TIE_BAR}s"),
    ("ʣ", f"d{TIE_BAR}z"),
    ("g", "ɡ"),
    ("'", PRIMARY_STRESS),
    (":", "ː"),
    (".", ""),  # syllable break
]

# Sequences read as a single phone even without a tie bar: affricates written
# this way in nearly every dictionary, and the English diphthongs. Anything else
# ("ts" in "cats" against Italian "pizza") is one phone only when tied.
_IPA_MULTI = ["tʃ", "dʒ", "aɪ", "aʊ", "ɔɪ", "oʊ", "eɪ"]

_IPA_VOWELS = set("aeiouyæɑɒɔəɚɛɜɝɪʊʌɐᵻɨʉɯɤøœɶɵɘɞ")

# Characters that belong to the phone before them rather than starting a new one.
_IPA_MODIFIERS = set("ːˑ")


[docs] def normalize_ipa(ipa: str) -> str: """Unify the notation of an IPA string without changing what it says. Strips surrounding ``/…/`` or ``[…]``, composes Unicode, replaces ligatures and ASCII look-alikes with their IPA symbols, writes ``tʃ`` and ``dʒ`` untied and other affricates tied (``ʦ`` becomes ``t͡s``), drops syllable breaks, and collapses whitespace. Args: ipa: An IPA transcription. Returns: The same transcription in canonical notation. Example:: normalize_ipa("/ˈʧɪ.kən/") # "ˈtʃɪkən" """ ipa = unicodedata.normalize("NFC", ipa.strip()) if len(ipa) >= 2 and (ipa[0], ipa[-1]) in (("/", "/"), ("[", "]")): ipa = ipa[1:-1] for old, new in _IPA_NOTATION: ipa = ipa.replace(old, new) return " ".join(ipa.split())
[docs] def ipa_phones(ipa: str) -> list[str]: """Split one IPA word into phones, in written order. Stress marks are returned as tokens of their own, where they were written. Tied sequences (``t͡s``), ``tʃ``, ``dʒ`` and the English diphthongs are one phone each, returned without the tie bar; length marks and combining diacritics stay with the phone they modify. Args: ipa: An IPA transcription of a single word. Returns: The phones and stress marks of the word. Example:: ipa_phones("mɛtˈfɔɹmɪn") # ["m", "ɛ", "t", "ˈ", "f", "ɔ", "ɹ", "m", "ɪ", "n"] """ ipa = normalize_ipa(ipa).replace(" ", "") phones: list[str] = [] i = 0 while i < len(ipa): char = ipa[i] if char in STRESS_MARKS: phones.append(char) i += 1 continue if ipa.startswith(TIE_BAR, i + 1) and i + 2 < len(ipa): phone = char + ipa[i + 2] i += 3 else: phone = next((m for m in _IPA_MULTI if ipa.startswith(m, i)), char) i += len(phone) while i < len(ipa) and (ipa[i] in _IPA_MODIFIERS or unicodedata.category(ipa[i]) == "Mn"): phone += ipa[i] i += 1 if phones and phones[-1] not in STRESS_MARKS and _is_modifier_only(phone): phones[-1] += phone else: phones.append(phone) return phones
[docs] def is_ipa_vowel(phone: str) -> bool: """Whether an IPA phone (as returned by :func:`ipa_phones`) is a vowel.""" return bool(phone) and phone[0] in _IPA_VOWELS
[docs] def stress_before_vowels(phones: list[str]) -> list[str]: """Move each stress mark from where it was written to just before its vowel. Transcriptions usually put stress at the start of the syllable (``mɛtˈfɔɹmɪn``); some services want it on the vowel itself (``mɛtfˈɔɹmɪn``). A stress mark with no vowel after it is dropped. Args: phones: Phones and stress marks, as returned by :func:`ipa_phones`. Returns: The same phones with every stress mark directly before a vowel. """ result: list[str] = [] pending: str | None = None for phone in phones: if phone in STRESS_MARKS: pending = phone continue if pending and is_ipa_vowel(phone): result.append(pending) pending = None result.append(phone) return result
def _is_modifier_only(phone: str) -> bool: return all(c in _IPA_MODIFIERS or unicodedata.category(c) == "Mn" for c in phone)