diff --git a/src/deepgram/helpers/text_builder.py b/src/deepgram/helpers/text_builder.py index b859b217..384b15eb 100644 --- a/src/deepgram/helpers/text_builder.py +++ b/src/deepgram/helpers/text_builder.py @@ -239,12 +239,19 @@ def ssml_to_deepgram(ssml_text: str) -> str: # Parse XML fragments manually to handle mixed content # Use regex to find and replace SSML elements - # Handle tags - phoneme_pattern = r'(.*?)' + # Handle tags. Attribute order is not significant in SSML, so match + # the attributes as a group and pull `ph` out of it rather than requiring + # `alphabet` before `ph` (which silently dropped the pronunciation otherwise). + phoneme_pattern = r"]*?)\s*>([^<]*)" def replace_phoneme(match): - ipa = match.group(1) + attributes = match.group(1) word = match.group(2) + alphabet_match = re.search(r'(?:^|\s)alphabet\s*=\s*(["\'])ipa\1(?=\s|$)', attributes) + ph_match = re.search(r'(?:^|\s)ph\s*=\s*(["\'])(.*?)\1(?=\s|$)', attributes) + if alphabet_match is None or ph_match is None: + return word + ipa = ph_match.group(2) return json.dumps({"word": word, "pronounce": ipa}, ensure_ascii=False) ssml_text = re.sub(phoneme_pattern, replace_phoneme, ssml_text) diff --git a/tests/custom/test_text_builder.py b/tests/custom/test_text_builder.py index 77a7ed1b..43c6270e 100644 --- a/tests/custom/test_text_builder.py +++ b/tests/custom/test_text_builder.py @@ -234,10 +234,52 @@ def test_basic_phoneme(self): """Test converting basic phoneme tag""" ssml = 'azathioprine' result = ssml_to_deepgram(ssml) - + assert '"word": "azathioprine"' in result assert '"pronounce": "ˌæzəˈθaɪəpriːn"' in result - + + def test_phoneme_attribute_order_independent(self): + """ph before alphabet must work too (SSML attribute order is not significant)""" + ssml = 'azathioprine' + result = ssml_to_deepgram(ssml) + + assert '"word": "azathioprine"' in result + assert '"pronounce": "ˌæzəˈθaɪəpriːn"' in result + + def test_phoneme_attributes_allow_whitespace_around_equals(self): + """Valid XML whitespace around attribute equals signs must be accepted""" + ssml = "azathioprine" + result = ssml_to_deepgram(ssml) + + assert '"word": "azathioprine"' in result + assert '"pronounce": "ˌæzəˈθaɪəpriːn"' in result + + @pytest.mark.parametrize( + "attributes", + [ + 'ph="test"', + 'alphabet="x-sampa" ph="test"', + 'alphabet="ipa" data-ph="test"', + ], + ) + def test_phoneme_requires_ipa_alphabet_and_ph_attribute(self, attributes): + """Unsupported or lookalike attributes must degrade to plain text""" + ssml = f"medicine" + + assert ssml_to_deepgram(ssml) == "medicine" + + def test_unclosed_phoneme_does_not_consume_following_phoneme(self): + """Malformed input must not capture a later valid phoneme tag""" + ssml = ( + '' + 'medicine' + ) + result = ssml_to_deepgram(ssml) + + assert '"word": "medicine"' in result + assert '"pronounce": "good"' in result + assert '"pronounce": "bad"' not in result + def test_basic_break(self): """Test converting break tag (milliseconds)""" ssml = '' @@ -497,4 +539,3 @@ def test_standalone_function_workflow(self): assert '"pronounce": "ˌæzəˈθaɪəpriːn"' in text assert '"word": "dupilumab"' in text assert '"pronounce": "duːˈpɪljuːmæb"' in text -