diff --git a/README.md b/README.md
index f5a185d..730548c 100644
--- a/README.md
+++ b/README.md
@@ -48,9 +48,24 @@ You can safely ignore Swift-specific files. Your respective package managers (Ca
| Apple | `"apple"` |
| W3C | `"w3c"` |
| Samsung Bixby | `"samsung-bixby"` or `"bixby"` |
-| ElevenLabs | `"elevenlabs"` |
+| ElevenLabs (pre-v3) | `"elevenlabs"` |
+| ElevenLabs v3 | `"elevenlabs-v3"` |
| IBM Watson | `"ibm-watson"` or `"watson"` |
+### ElevenLabs note
+
+ElevenLabs does not parse SSML documents, so both ElevenLabs platforms emit
+prompt markup rather than SSML, and the correct dialect depends on the model:
+
+- `"elevenlabs"` — pre-v3 models (`eleven_multilingual_v2`, `eleven_flash_v2_5`,
+ `eleven_flash_v2`, `eleven_turbo_v2`): `` pauses (max 3s)
+ and `` for IPA (flash/turbo, English only). No `` wrapper.
+- `"elevenlabs-v3"` — `eleven_v3` / `eleven_v3_conversational`: bracketed
+ natural-language audio tags (`[whispers]`, `[excited]`, `[pause]`,
+ `[long pause]`, `[laughs]`), punctuation pauses, and native `"/IPA/"`.
+ Audio tags are model-interpreted direction — best-effort, voice-dependent —
+ and are not understood by pre-v3 models.
+
## API
All bindings expose the same core methods:
diff --git a/bindings/dotnet/SpeechMarkdown.cs b/bindings/dotnet/SpeechMarkdown.cs
index d1c5840..c149fa6 100644
--- a/bindings/dotnet/SpeechMarkdown.cs
+++ b/bindings/dotnet/SpeechMarkdown.cs
@@ -274,7 +274,9 @@ public static class Platform
public const string Apple = "apple";
public const string W3c = "w3c";
public const string SamsungBixby = "samsung-bixby";
- public const string ElevenLabs = "eleven-labs";
+ // Note: the core library parses "elevenlabs" (no hyphen).
+ public const string ElevenLabs = "elevenlabs";
+ public const string ElevenLabsV3 = "elevenlabs-v3";
public const string IbmWatson = "ibm-watson";
}
}
diff --git a/bindings/nodejs/src/lib.rs b/bindings/nodejs/src/lib.rs
index fbbaed3..3def523 100644
--- a/bindings/nodejs/src/lib.rs
+++ b/bindings/nodejs/src/lib.rs
@@ -5,7 +5,7 @@ use speechmarkdown_rust::{Platform, SpeechMarkdownParser};
fn parse_platform(platform: &str) -> Result {
Platform::from_platform_str(platform).ok_or_else(|| {
Error::from_reason(format!(
- "unsupported platform: '{}'. Use one of: amazon-alexa, google-assistant, microsoft-azure, apple, w3c, samsung-bixby, elevenlabs, ibm-watson",
+ "unsupported platform: '{}'. Use one of: amazon-alexa, google-assistant, microsoft-azure, apple, w3c, samsung-bixby, elevenlabs, elevenlabs-v3, ibm-watson",
platform
))
})
diff --git a/bindings/python/src/lib.rs b/bindings/python/src/lib.rs
index 79172f4..4763e60 100644
--- a/bindings/python/src/lib.rs
+++ b/bindings/python/src/lib.rs
@@ -4,7 +4,7 @@ use speechmarkdown_rust::{Platform, SpeechMarkdownParser};
fn parse_platform(platform: &str) -> PyResult {
Platform::from_platform_str(platform).ok_or_else(|| {
pyo3::exceptions::PyValueError::new_err(format!(
- "unsupported platform: '{}'. Use one of: amazon-alexa, google-assistant, microsoft-azure, apple, w3c, samsung-bixby, elevenlabs, ibm-watson",
+ "unsupported platform: '{}'. Use one of: amazon-alexa, google-assistant, microsoft-azure, apple, w3c, samsung-bixby, elevenlabs, elevenlabs-v3, ibm-watson",
platform
))
})
diff --git a/bindings/swift/Sources/CSpeechMarkdown/include/speechmarkdown.h b/bindings/swift/Sources/CSpeechMarkdown/include/speechmarkdown.h
index d2e1da1..d84bee6 100644
--- a/bindings/swift/Sources/CSpeechMarkdown/include/speechmarkdown.h
+++ b/bindings/swift/Sources/CSpeechMarkdown/include/speechmarkdown.h
@@ -10,7 +10,7 @@ extern "C" {
// Convert SpeechMarkdown input to SSML for the given platform.
// Platforms: "amazon-alexa", "google-assistant", "microsoft-azure",
-// "apple", "w3c", "samsung-bixby", "eleven-labs", "ibm-watson"
+// "apple", "w3c", "samsung-bixby", "elevenlabs", "elevenlabs-v3", "ibm-watson"
// Returns: allocated string with SSML, or NULL on error.
// Caller must free with speechmarkdown_free().
const char* speechmarkdown_to_ssml(const char* input, const char* platform);
diff --git a/src/ast.rs b/src/ast.rs
index 6134637..a875e81 100644
--- a/src/ast.rs
+++ b/src/ast.rs
@@ -162,6 +162,10 @@ pub enum NodeType {
/// Mark tag
Mark,
+ /// Expressive audio tag [laugh], [sigh], [applause], etc.
+ /// Kept verbatim by formatters as a bracketed audio tag.
+ Expressive,
+
// Modifier types (for text modifiers and sections)
/// Emphasis modifier
Emphasis,
diff --git a/src/capabilities.rs b/src/capabilities.rs
index b3bdbd2..6e5751d 100644
--- a/src/capabilities.rs
+++ b/src/capabilities.rs
@@ -26,6 +26,7 @@ pub fn get_supported_ssml(platform: Platform) -> PlatformCapabilities {
Platform::W3c => w3c_capabilities(),
Platform::SamsungBixby => samsung_bixby_capabilities(),
Platform::ElevenLabs => elevenlabs_capabilities(),
+ Platform::ElevenLabsV3 => elevenlabs_v3_capabilities(),
Platform::IbmWatson => ibm_watson_capabilities(),
}
}
@@ -396,24 +397,127 @@ fn samsung_bixby_capabilities() -> PlatformCapabilities {
}
fn elevenlabs_capabilities() -> PlatformCapabilities {
+ // Pre-v3 prompt markup (eleven_multilingual_v2 / flash_v2_5 /
+ // flash_v2 / turbo_v2): no SSML document, but two inline tags are
+ // parsed. Everything else degrades to plain text.
PlatformCapabilities {
platform: "elevenlabs".into(),
ssml_elements: vec![
- break_element(),
- prosody_element(),
- audio_element(),
- phoneme_element(),
- mark_element(),
- say_as_element("characters", "characters"),
- say_as_element("number", "number"),
- say_as_element("date", "date"),
- say_as_element("time", "time"),
+ SsmlCapability {
+ element: "break".into(),
+ description: "Insert a pause (pre-v3 models only; max 3s; \
+ eleven_v3 does not parse break tags)"
+ .into(),
+ attributes: vec!["time".into()],
+ speech_markdown_syntax: vec![
+ "[2s]".into(),
+ "[500ms]".into(),
+ "[break:strong]".into(),
+ ],
+ example: "Hello [2s] world".into(),
+ },
+ SsmlCapability {
+ element: "phoneme".into(),
+ description: "Custom pronunciation (eleven_flash_v2 / \
+ eleven_turbo_v2 only, English only)"
+ .into(),
+ attributes: vec!["alphabet".into(), "ph".into()],
+ speech_markdown_syntax: vec![
+ "(text)[ipa:\"pɪkəloʊ\"]".into(),
+ "(text)/pɪkəloʊ/".into(),
+ ],
+ example: "(piccolo)/pɪkəloʊ/".into(),
+ },
],
unsupported: vec![
- "emphasis".into(),
- "voice".into(),
- "lang".into(),
- "sub".into(),
+ "emphasis (no equivalent; degrades to text)".into(),
+ "prosody (rate maps to the API voice_settings.speed; use the \
+ elevenlabs-v3 dialect for tag-based control)"
+ .into(),
+ "say-as (text normalization is built in)".into(),
+ "sub (use pronunciation dictionaries or phonetic spelling)".into(),
+ "audio (no equivalent)".into(),
+ "mark (no equivalent)".into(),
+ "voice (switch via the API voice_id parameter)".into(),
+ "lang (no equivalent)".into(),
+ "audio tags (eleven_v3 only)".into(),
+ "amazon:effect".into(),
+ "amazon:emotion".into(),
+ "amazon:domain".into(),
+ "mstts:express-as".into(),
+ "google:style".into(),
+ ],
+ }
+}
+
+fn elevenlabs_v3_capabilities() -> PlatformCapabilities {
+ // Eleven v3 audio-tag dialect: no SSML at all. Bracketed
+ // natural-language tags are prompts interpreted by the model —
+ // best-effort, voice-dependent, and directional from their insertion
+ // point (no span semantics).
+ PlatformCapabilities {
+ platform: "elevenlabs-v3".into(),
+ ssml_elements: vec![
+ SsmlCapability {
+ element: "[audio tag]".into(),
+ description: "Natural-language performance direction in \
+ brackets; open-ended (emotions, delivery, \
+ reactions, sound effects, accents)"
+ .into(),
+ attributes: vec![],
+ speech_markdown_syntax: vec![
+ "[laugh]".into(),
+ "#[excited] text".into(),
+ "(text)[whisper]".into(),
+ ],
+ example: "[whispers] It's a secret".into(),
+ },
+ SsmlCapability {
+ element: "[pause] family".into(),
+ description: "Pauses via [short pause] / [pause] / [long \
+ pause] or ellipses; approximate — no exact \
+ durations on v3"
+ .into(),
+ attributes: vec![],
+ speech_markdown_syntax: vec![
+ "[500ms]".into(),
+ "[2s]".into(),
+ "[break:strong]".into(),
+ ],
+ example: "Hello [2s] world".into(),
+ },
+ SsmlCapability {
+ element: "inline IPA".into(),
+ description: "Native IPA wrapped in quotes and slashes".into(),
+ attributes: vec![],
+ speech_markdown_syntax: vec!["(speech)/spitʃ/".into()],
+ example: "(speech)/spitʃ/".into(),
+ },
+ SsmlCapability {
+ element: "emphasis tags".into(),
+ description: "Emphasis via [emphasized] / [stress on next \
+ word] / [understated] (capitalization also \
+ works but mutates the text)"
+ .into(),
+ attributes: vec![],
+ speech_markdown_syntax: vec!["++word++".into(), "+word+".into()],
+ example: "++important++".into(),
+ },
+ ],
+ unsupported: vec![
+ "break (eleven_v3 does not parse SSML break tags)".into(),
+ "phoneme (use inline IPA)".into(),
+ "prosody (rate maps to the API voice_settings.speed; \
+ tempo/volume map to [rushed]/[drawn out]/[softly]/[loudly] tags)"
+ .into(),
+ "say-as (text normalization is built in)".into(),
+ "sub (alias is spoken instead of the text)".into(),
+ "audio (no equivalent)".into(),
+ "mark (no equivalent)".into(),
+ "voice (switch via the API voice_id parameter or Text to \
+ Dialogue)"
+ .into(),
+ "lang (no equivalent)".into(),
"amazon:effect".into(),
"amazon:emotion".into(),
"amazon:domain".into(),
@@ -466,24 +570,64 @@ mod tests {
Platform::W3c,
Platform::SamsungBixby,
Platform::ElevenLabs,
+ Platform::ElevenLabsV3,
Platform::IbmWatson,
] {
let caps = get_supported_ssml(platform);
- assert!(!caps.ssml_elements.is_empty(), "{:?} has no elements", platform);
+ assert!(
+ !caps.ssml_elements.is_empty(),
+ "{:?} has no elements",
+ platform
+ );
assert!(!caps.platform.is_empty());
}
}
+ #[test]
+ fn test_all_platforms_have_break() {
+ for platform in [
+ Platform::AmazonAlexa,
+ Platform::GoogleAssistant,
+ Platform::MicrosoftAzure,
+ Platform::Apple,
+ Platform::W3c,
+ Platform::SamsungBixby,
+ Platform::ElevenLabs,
+ Platform::IbmWatson,
+ ] {
+ let caps = get_supported_ssml(platform);
+ assert!(
+ caps.ssml_elements.iter().any(|e| e.element == "break"),
+ "{:?} missing break",
+ platform
+ );
+ }
+
+ // Eleven v3 does not parse at all; pauses are audio tags.
+ let v3 = get_supported_ssml(Platform::ElevenLabsV3);
+ assert!(v3
+ .ssml_elements
+ .iter()
+ .any(|e| e.element == "[pause] family"));
+ assert!(v3.unsupported.iter().any(|u| u.starts_with("break")));
+ }
+
#[test]
fn test_alexa_has_emotion() {
let caps = get_supported_ssml(Platform::AmazonAlexa);
- assert!(caps.ssml_elements.iter().any(|e| e.element == "amazon:emotion"));
+ assert!(caps
+ .ssml_elements
+ .iter()
+ .any(|e| e.element == "amazon:emotion"));
}
#[test]
fn test_azure_has_express_as() {
let caps = get_supported_ssml(Platform::MicrosoftAzure);
- assert!(caps.ssml_elements.iter().any(|e| e.element == "mstts:express-as"));
+ assert!(caps
+ .ssml_elements
+ .iter()
+ .any(|e| e.element == "mstts:express-as"));
}
#[test]
@@ -506,25 +650,4 @@ mod tests {
let deserialized: PlatformCapabilities = serde_json::from_str(&json).unwrap();
assert_eq!(caps, deserialized);
}
-
- #[test]
- fn test_all_platforms_have_break() {
- for platform in [
- Platform::AmazonAlexa,
- Platform::GoogleAssistant,
- Platform::MicrosoftAzure,
- Platform::Apple,
- Platform::W3c,
- Platform::SamsungBixby,
- Platform::ElevenLabs,
- Platform::IbmWatson,
- ] {
- let caps = get_supported_ssml(platform);
- assert!(
- caps.ssml_elements.iter().any(|e| e.element == "break"),
- "{:?} missing break",
- platform
- );
- }
- }
}
diff --git a/src/formatters/base.rs b/src/formatters/base.rs
index 10b5a40..a50cf5c 100644
--- a/src/formatters/base.rs
+++ b/src/formatters/base.rs
@@ -12,6 +12,9 @@ pub enum Platform {
W3c,
SamsungBixby,
ElevenLabs,
+ /// Eleven v3 audio-tag dialect (eleven_v3 / eleven_v3_conversational):
+ /// bracketed natural-language tags instead of SSML.
+ ElevenLabsV3,
IbmWatson,
}
@@ -26,6 +29,7 @@ impl Platform {
"w3c" => Some(Platform::W3c),
"samsung-bixby" | "bixby" => Some(Platform::SamsungBixby),
"elevenlabs" => Some(Platform::ElevenLabs),
+ "elevenlabs-v3" | "elevenlabs_v3" | "eleven-v3" => Some(Platform::ElevenLabsV3),
"ibm-watson" | "watson" => Some(Platform::IbmWatson),
_ => None,
}
@@ -41,6 +45,7 @@ impl Platform {
Platform::W3c => "w3c",
Platform::SamsungBixby => "samsung-bixby",
Platform::ElevenLabs => "elevenlabs",
+ Platform::ElevenLabsV3 => "elevenlabs-v3",
Platform::IbmWatson => "ibm-watson",
}
}
@@ -105,6 +110,16 @@ pub fn create_formatter(platform: Platform, options: FormatterOptions) -> Box {
+ // Pre-v3 prompt markup: tags (and flash-only ),
+ // no SSML document, no escaping.
+ Box::new(super::elevenlabs::ElevenLabsFormatter::new(options))
+ }
+ Platform::ElevenLabsV3 => {
+ // Eleven v3 audio-tag dialect: no SSML at all — bracketed
+ // natural-language tags, punctuation pauses, native slash IPA.
+ Box::new(super::elevenlabs_v3::ElevenLabsV3Formatter::new(options))
+ }
_ => Box::new(super::TextFormatter::new()),
}
}
diff --git a/src/formatters/elevenlabs.rs b/src/formatters/elevenlabs.rs
new file mode 100644
index 0000000..53fea1e
--- /dev/null
+++ b/src/formatters/elevenlabs.rs
@@ -0,0 +1,370 @@
+use crate::ast::{AstNode, NodeType};
+use crate::error::Result;
+use crate::formatters::base::{Formatter, FormatterOptions};
+
+/// Break strength → approximate duration, matching the speechmarkdown-js
+/// ElevenLabsFormatter reference. ElevenLabs only accepts `time` (no
+/// `strength` attribute), up to 3 seconds.
+const BREAK_STRENGTH_TO_DURATION: &[(&str, &str)] = &[
+ ("none", "0s"),
+ ("x-weak", "0.2s"),
+ ("weak", "0.35s"),
+ ("medium", "0.5s"),
+ ("strong", "0.8s"),
+ ("x-strong", "1.2s"),
+];
+
+const DEFAULT_BREAK_DURATION: &str = "0.5s";
+
+/// ElevenLabs pre-v3 models accept pauses up to 3 seconds; longer
+/// requested breaks are clamped (values beyond that are rejected or
+/// destabilize the generation).
+const MAX_BREAK_SECONDS: f64 = 3.0;
+
+/// ElevenLabs prompt markup for the pre-v3 model family
+/// (`eleven_multilingual_v2`, `eleven_flash_v2_5`, `eleven_flash_v2`,
+/// `eleven_turbo_v2`).
+///
+/// These models do not parse SSML documents, but they do understand two
+/// inline XML-style tags:
+///
+/// - `` — exact pause, up to 3 seconds, seconds format
+/// - `` — pronunciation control,
+/// `eleven_flash_v2` / `eleven_turbo_v2` only, English only
+///
+/// Everything else in SpeechMarkdown degrades to plain text. Output is a
+/// bare prompt: no `` wrapper and no XML escaping (matching the
+/// speechmarkdown-js reference formatter and the shared test corpus).
+///
+/// `eleven_v3` models must use the audio-tag dialect instead
+/// (`Platform::ElevenLabsV3`): they do not parse `` at all.
+pub struct ElevenLabsFormatter {
+ preserve_empty_lines: bool,
+}
+
+impl ElevenLabsFormatter {
+ pub fn new(options: FormatterOptions) -> Self {
+ Self {
+ preserve_empty_lines: options.preserve_empty_lines,
+ }
+ }
+
+ /// Parse a duration like "2", "0.25", "250ms", "1.5s" into seconds.
+ fn parse_seconds(text: &str) -> Option {
+ let body = text
+ .strip_suffix("ms")
+ .or_else(|| text.strip_suffix('s'))
+ .unwrap_or(text);
+ let value: f64 = body.parse().ok()?;
+ let scale = if text.ends_with("ms") { 0.001 } else { 1.0 };
+ Some(value * scale)
+ }
+
+ /// Break time verbatim when within the documented 3s limit; clamped
+ /// to 3s beyond it (keeping the caller's unit formatting otherwise,
+ /// matching the shared corpus fixtures which use both "3s" and "250ms").
+ fn clamp_break_time(time: &str) -> String {
+ match Self::parse_seconds(time) {
+ Some(secs) if secs > MAX_BREAK_SECONDS => format!("{MAX_BREAK_SECONDS}s"),
+ _ => time.to_string(),
+ }
+ }
+
+ fn map_strength_to_time(strength: &str) -> String {
+ let normalized = strength.trim().to_lowercase();
+ BREAK_STRENGTH_TO_DURATION
+ .iter()
+ .find(|(k, _)| *k == normalized)
+ .map(|(_, v)| (*v).to_string())
+ .unwrap_or_else(|| DEFAULT_BREAK_DURATION.to_string())
+ }
+
+ /// Strip the `[` / `]` the parser keeps in `ShortBreak::text`
+ /// (e.g. "[2s]" → "2s").
+ fn break_time_from_text(text: &str) -> &str {
+ text.trim_start_matches('[').trim_end_matches(']')
+ }
+
+ /// Escape a double-quoted attribute value. The prompt body stays
+ /// unescaped (ElevenLabs is not an XML document), but attribute
+ /// quotes would break the tag itself.
+ fn escape_attr(value: &str) -> String {
+ value.replace('"', """)
+ }
+
+ fn format_node_internal(&self, node: &AstNode, out: &mut String) -> Result<()> {
+ match node.node_type {
+ NodeType::Document => {
+ for child in &node.children {
+ self.format_node_internal(child, out)?;
+ }
+ }
+
+ // Sections have no pre-v3 equivalent: the marker itself is
+ // dropped and the section's content flows as plain text.
+ NodeType::Section => {}
+
+ NodeType::PlainText
+ | NodeType::PlainTextSpecialChars
+ | NodeType::PlainTextEmphasis
+ | NodeType::SimpleLine
+ | NodeType::Paragraph => {
+ out.push_str(&node.text);
+ for child in &node.children {
+ self.format_node_internal(child, out)?;
+ }
+ }
+
+ NodeType::EmptyLine => {
+ if self.preserve_empty_lines {
+ out.push('\n');
+ }
+ }
+
+ NodeType::ShortBreak => {
+ let time = Self::clamp_break_time(Self::break_time_from_text(&node.text));
+ out.push_str(&format!("", time));
+ }
+
+ NodeType::Break => {
+ let strength = node
+ .attributes
+ .get("strength")
+ .unwrap_or(&node.text)
+ .clone();
+ let time = Self::map_strength_to_time(&strength);
+ out.push_str(&format!("", time));
+ }
+
+ NodeType::ShortEmphasisModerate
+ | NodeType::ShortEmphasisStrong
+ | NodeType::ShortEmphasisNone
+ | NodeType::ShortEmphasisReduced => {
+ out.push_str(&node.text);
+ }
+
+ NodeType::TextModifier => {
+ let phoneme = node.attributes.get("ipa");
+ if let Some(ph) = phoneme.filter(|ph| !ph.is_empty()) {
+ out.push_str(&format!(
+ "{}",
+ Self::escape_attr(ph),
+ node.text
+ ));
+ } else {
+ out.push_str(&node.text);
+ }
+ }
+
+ NodeType::ShortIpa => {
+ let phoneme = node.attributes.get("phoneme");
+ if let Some(ph) = phoneme.filter(|ph| !ph.is_empty()) {
+ out.push_str(&format!(
+ "{}",
+ Self::escape_attr(ph),
+ node.text
+ ));
+ } else {
+ out.push_str(&node.text);
+ }
+ }
+
+ NodeType::BareIpa => {
+ // A bare `/ipa/` has no display word; keep the phoneme
+ // characters as text (reference-formatter behavior).
+ if let Some(ph) = node.attributes.get("ph") {
+ out.push_str(ph);
+ } else {
+ out.push_str(&node.text);
+ }
+ }
+
+ NodeType::ShortSub => {
+ out.push_str(&node.text);
+ }
+
+ // No pre-v3 equivalent: drop rather than speak a URL or
+ // emit markup the model would read aloud.
+ NodeType::Audio | NodeType::Mark => {}
+
+ NodeType::Expressive => {
+ out.push_str(&format!("[{}]", node.text));
+ }
+
+ // Modifier node types never appear standalone from the parser.
+ _ => {}
+ }
+ Ok(())
+ }
+}
+
+impl Formatter for ElevenLabsFormatter {
+ fn format(&self, ast: &AstNode) -> Result {
+ let mut out = String::new();
+ self.format_node_internal(ast, &mut out)?;
+ Ok(out)
+ }
+
+ fn format_node(&self, node: &AstNode) -> Result {
+ let mut out = String::new();
+ self.format_node_internal(node, &mut out)?;
+ Ok(out)
+ }
+}
+
+#[cfg(test)]
+mod tests {
+ use crate::formatters::base::Platform;
+ use crate::parser::SpeechMarkdownParser;
+
+ fn to_elevenlabs(input: &str) -> String {
+ SpeechMarkdownParser::to_ssml(input, Platform::ElevenLabs).unwrap()
+ }
+
+ #[test]
+ fn short_break_becomes_break_tag() {
+ assert_eq!(
+ to_elevenlabs("Sample [3s] speech [250ms] markdown"),
+ "Sample speech markdown"
+ );
+ }
+
+ #[test]
+ fn quoted_break_time_becomes_break_tag() {
+ assert_eq!(
+ to_elevenlabs("Sample [break:\"3s\"] speech [break:'250ms'] markdown"),
+ "Sample speech markdown"
+ );
+ }
+
+ #[test]
+ fn break_strength_maps_to_duration() {
+ assert_eq!(
+ to_elevenlabs("[break:\"medium\"]"),
+ ""
+ );
+ assert_eq!(
+ to_elevenlabs("[break:\"x-strong\"]"),
+ ""
+ );
+ assert_eq!(to_elevenlabs("[break:\"none\"]"), "");
+ // Unknown strength falls back to medium duration.
+ assert_eq!(to_elevenlabs("[break:\"bogus\"]"), "");
+ }
+
+ #[test]
+ fn breaks_clamp_to_three_seconds() {
+ // Pre-v3 models accept at most 3s; longer values are rejected
+ // or destabilize the generation.
+ assert_eq!(
+ to_elevenlabs("Wait [10s] now"),
+ "Wait now"
+ );
+ assert_eq!(
+ to_elevenlabs("Wait [3500ms] now"),
+ "Wait now"
+ );
+ // Values within the limit keep the caller's unit formatting
+ // (the corpus fixtures pin both "3s" and "250ms").
+ assert_eq!(
+ to_elevenlabs("Wait [250ms] now"),
+ "Wait now"
+ );
+ }
+
+ #[test]
+ fn malformed_break_numbers_are_not_breaks() {
+ // Not valid durations: plain text passthrough (matches the
+ // speechmarkdown-js grammar, which only accepts \d+(\.\d+)?(s|ms)).
+ for word in [
+ "1.2.3s",
+ "1..5s",
+ "apps",
+ "infs",
+ "s",
+ "5.s",
+ ".5s",
+ "0.s",
+ "-2s",
+ "+2s",
+ "1e3s",
+ ] {
+ let out = to_elevenlabs(&format!("x [{word}] y"));
+ assert_eq!(out, format!("x [{word}] y"), "word {word}");
+ }
+ }
+
+ #[test]
+ fn phoneme_attribute_quotes_are_escaped() {
+ assert_eq!(
+ to_elevenlabs("(x)[ipa:\"a\"b\"]"),
+ "x"
+ );
+ }
+
+ #[test]
+ fn no_speak_wrapper_and_no_escaping() {
+ assert_eq!(
+ to_elevenlabs("1 < 2 & 3 > 0 \"yes\" 'no'"),
+ "1 < 2 & 3 > 0 \"yes\" 'no'"
+ );
+ }
+
+ #[test]
+ fn empty_modifier_list_strips_to_text() {
+ assert_eq!(to_elevenlabs("Some (text)[]"), "Some text");
+ }
+
+ #[test]
+ fn unsupported_modifiers_degrade_to_text() {
+ assert_eq!(
+ to_elevenlabs("(read this)[rate:\"fast\";volume:\"loud\"]"),
+ "read this"
+ );
+ assert_eq!(to_elevenlabs("++important++"), "important");
+ assert_eq!(to_elevenlabs("(hello)[voice:\"Brian\"]"), "hello");
+ }
+
+ #[test]
+ fn ipa_modifier_emits_phoneme_tag() {
+ assert_eq!(
+ to_elevenlabs("(piccolo)[ipa:\"pɪkəloʊ\"]"),
+ "piccolo"
+ );
+ }
+
+ #[test]
+ fn short_ipa_emits_phoneme_tag() {
+ assert_eq!(
+ to_elevenlabs("(speech)/spitʃ/"),
+ "speech"
+ );
+ }
+
+ #[test]
+ fn expressive_tags_pass_through() {
+ assert_eq!(
+ to_elevenlabs("He [laugh] and then [applause] left"),
+ "He [laugh] and then [applause] left"
+ );
+ }
+
+ #[test]
+ fn sections_are_dropped_content_flows() {
+ assert_eq!(to_elevenlabs("#[excited] Hello world"), " Hello world");
+ }
+
+ #[test]
+ fn sub_keeps_display_text() {
+ assert_eq!(to_elevenlabs("{AL}aluminum"), "AL");
+ }
+
+ #[test]
+ fn audio_and_mark_are_dropped() {
+ // Dropped nodes leave the surrounding spacing in place.
+ assert_eq!(
+ to_elevenlabs("Hello [mark:chapter1]  world"),
+ "Hello world"
+ );
+ }
+}
diff --git a/src/formatters/elevenlabs_v3.rs b/src/formatters/elevenlabs_v3.rs
new file mode 100644
index 0000000..800965c
--- /dev/null
+++ b/src/formatters/elevenlabs_v3.rs
@@ -0,0 +1,399 @@
+use crate::ast::{AstNode, NodeType};
+use crate::error::Result;
+use crate::formatters::base::{Formatter, FormatterOptions};
+
+/// Eleven v3 audio-tag dialect for `eleven_v3` / `eleven_v3_conversational`.
+///
+/// Eleven v3 does not parse SSML — no ``, no ``. Delivery is
+/// directed with bracketed natural-language audio tags (`[whispers]`,
+/// `[laughs]`, `[pause]`), punctuation (`...`, em-dash) and capitalization.
+/// Tags are open-ended prompts interpreted by the model, not an enum: they
+/// are best-effort and voice-dependent, and they directionally apply from
+/// their insertion point onward (there is no guaranteed "end tag" span).
+///
+/// Mapping notes:
+/// - Breaks lose temporal precision: three steps plus punctuation
+/// (`...` / `[short pause]` / `[pause]` / `[long pause]`).
+/// - Emphasis avoids mutating the user's words (no CAPS by default).
+/// - IPA is emitted in v3's native `"/…/"` slash form.
+/// - `#[style]` sections become prefix tags; unknown styles pass through
+/// verbatim (v3 treats any bracketed cue as direction).
+pub struct ElevenLabsV3Formatter {
+ preserve_empty_lines: bool,
+}
+
+/// Break strength → v3 pause tag. No temporal precision available.
+const BREAK_STRENGTH_TO_TAG: &[(&str, &str)] = &[
+ ("none", ""),
+ ("x-weak", "..."),
+ ("weak", "[short pause]"),
+ ("medium", "[pause]"),
+ ("strong", "[long pause]"),
+ ("x-strong", "[long pause]"),
+];
+
+const DEFAULT_PAUSE_TAG: &str = "[pause]";
+
+impl ElevenLabsV3Formatter {
+ pub fn new(options: FormatterOptions) -> Self {
+ Self {
+ preserve_empty_lines: options.preserve_empty_lines,
+ }
+ }
+
+ /// Parse "2", "0.25", "250ms", "1.5s", … into seconds.
+ fn parse_seconds(text: &str) -> Option {
+ let body = text
+ .strip_suffix("ms")
+ .or_else(|| text.strip_suffix('s'))
+ .unwrap_or(text);
+ let value: f64 = body.parse().ok()?;
+ let scale = if text.ends_with("ms") { 0.001 } else { 1.0 };
+ Some(value * scale)
+ }
+
+ fn pause_tag_for_seconds(secs: f64) -> &'static str {
+ if secs < 0.4 {
+ "..."
+ } else if secs <= 1.0 {
+ "[pause]"
+ } else {
+ "[long pause]"
+ }
+ }
+
+ fn strength_to_tag(strength: &str) -> &'static str {
+ let normalized = strength.trim().to_lowercase();
+ BREAK_STRENGTH_TO_TAG
+ .iter()
+ .find(|(k, _)| *k == normalized)
+ .map(|(_, v)| *v)
+ .unwrap_or(DEFAULT_PAUSE_TAG)
+ }
+
+ fn break_time_from_text(text: &str) -> &str {
+ text.trim_start_matches('[').trim_end_matches(']')
+ }
+
+ /// Map a modifier key/value pair to a prefix audio tag (or None to
+ /// degrade to plain text). IPA is handled separately by the caller
+ /// because it replaces the text instead of prefixing it.
+ fn modifier_to_tag(key: &str, value: &str) -> Option {
+ let value = value.trim().to_lowercase();
+ match key.to_lowercase().as_str() {
+ "whisper" => Some("[whispers]".to_string()),
+ "excited" => Some("[excited]".to_string()),
+ "disappointed" => Some("[disappointed]".to_string()),
+ "rate" => match value.as_str() {
+ "x-slow" | "slow" => Some("[drawn out]".to_string()),
+ "fast" | "x-fast" => Some("[rushed]".to_string()),
+ _ => None,
+ },
+ "volume" | "vol" => match value.as_str() {
+ "x-soft" | "soft" | "quiet" => Some("[softly]".to_string()),
+ "x-loud" | "loud" => Some("[loudly]".to_string()),
+ _ => None,
+ },
+ "emphasis" => match value.as_str() {
+ "strong" => Some("[emphasized]".to_string()),
+ "moderate" => Some("[stress on next word]".to_string()),
+ "reduced" => Some("[understated]".to_string()),
+ _ => None,
+ },
+ _ => None,
+ }
+ }
+
+ fn format_node_internal(&self, node: &AstNode, out: &mut String) -> Result<()> {
+ match node.node_type {
+ NodeType::Document => {
+ // Sections prefix the content that follows them (until the
+ // next section); there is no closing form on v3.
+ let mut iter = node.children.iter().peekable();
+ while let Some(child) = iter.next() {
+ if child.node_type == NodeType::Section {
+ let mut section_content = String::new();
+ while let Some(next) = iter.peek() {
+ if next.node_type == NodeType::Section {
+ break;
+ }
+ let next = iter.next().unwrap();
+ self.format_node_internal(next, &mut section_content)?;
+ }
+ out.push_str(&self.section_prefix(child));
+ out.push_str(§ion_content);
+ } else {
+ self.format_node_internal(child, out)?;
+ }
+ }
+ }
+
+ NodeType::PlainText
+ | NodeType::PlainTextSpecialChars
+ | NodeType::PlainTextEmphasis
+ | NodeType::SimpleLine
+ | NodeType::Paragraph => {
+ out.push_str(&node.text);
+ for child in &node.children {
+ self.format_node_internal(child, out)?;
+ }
+ }
+
+ NodeType::EmptyLine => {
+ if self.preserve_empty_lines {
+ out.push('\n');
+ }
+ }
+
+ NodeType::ShortBreak => {
+ let time = Self::break_time_from_text(&node.text);
+ let tag = Self::parse_seconds(time)
+ .map(Self::pause_tag_for_seconds)
+ .unwrap_or(DEFAULT_PAUSE_TAG);
+ out.push_str(tag);
+ }
+
+ NodeType::Break => {
+ let strength = node
+ .attributes
+ .get("strength")
+ .unwrap_or(&node.text)
+ .clone();
+ out.push_str(Self::strength_to_tag(&strength));
+ }
+
+ NodeType::ShortEmphasisStrong => {
+ out.push_str("[emphasized] ");
+ out.push_str(&node.text);
+ }
+ NodeType::ShortEmphasisModerate => {
+ out.push_str("[stress on next word] ");
+ out.push_str(&node.text);
+ }
+ NodeType::ShortEmphasisReduced => {
+ out.push_str("[understated] ");
+ out.push_str(&node.text);
+ }
+ NodeType::ShortEmphasisNone => {
+ out.push_str(&node.text);
+ }
+
+ NodeType::TextModifier => {
+ // IPA replaces the text; other recognized modifiers
+ // become prefix tags in declaration order.
+ if let Some(ph) = node.attributes.get("ipa").filter(|v| !v.is_empty()) {
+ out.push_str(&format!("\"/{}/\"", ph));
+ } else if let Some(alias) = node.attributes.get("sub").filter(|v| !v.is_empty()) {
+ // Substitution: speak the alias instead of the text.
+ out.push_str(alias);
+ } else {
+ let mut tags: Vec = Vec::new();
+ for key in &node.attribute_keys {
+ let value = node.attributes.get(key).map(String::as_str).unwrap_or("");
+ if let Some(tag) = Self::modifier_to_tag(key, value) {
+ tags.push(tag);
+ }
+ }
+ for tag in &tags {
+ out.push_str(tag);
+ out.push(' ');
+ }
+ out.push_str(&node.text);
+ }
+ }
+
+ NodeType::ShortIpa => {
+ // v3 native IPA replaces the word with "/phoneme/".
+ if let Some(ph) = node.attributes.get("phoneme").filter(|v| !v.is_empty()) {
+ out.push_str(&format!("\"/{}/\"", ph));
+ } else {
+ out.push_str(&node.text);
+ }
+ }
+
+ NodeType::BareIpa => {
+ if let Some(ph) = node.attributes.get("ph") {
+ out.push_str(&format!("\"/{}/\"", ph));
+ } else {
+ out.push_str(&node.text);
+ }
+ }
+
+ NodeType::ShortSub => {
+ // Speak the alias (the intended spoken form) when present.
+ if let Some(alias) = node.attributes.get("alias").filter(|v| !v.is_empty()) {
+ out.push_str(alias);
+ } else {
+ out.push_str(&node.text);
+ }
+ }
+
+ NodeType::Audio | NodeType::Mark => {}
+
+ NodeType::Expressive => {
+ out.push_str(&format!("[{}]", node.text));
+ }
+
+ NodeType::Section => {
+ // Handled by the document walk; standalone formatting of a
+ // section still emits its prefix tags.
+ out.push_str(&self.section_prefix(node));
+ }
+
+ // Modifier node types never appear standalone from the parser.
+ _ => {}
+ }
+ Ok(())
+ }
+
+ /// Prefix tags for a `#[…]` section: the bare style passes through as a
+ /// natural-language tag; recognized modifier keys map to tempo/volume
+ /// tags; everything else is dropped (no v3 equivalent).
+ fn section_prefix(&self, node: &AstNode) -> String {
+ let mut tags: Vec = Vec::new();
+
+ if let Some(style) = node.attributes.get("style") {
+ if style != "defaults" && !style.is_empty() {
+ tags.push(format!("[{}]", style));
+ }
+ }
+
+ for key in &node.attribute_keys {
+ if key == "style" {
+ continue;
+ }
+ let value = node.attributes.get(key).map(String::as_str).unwrap_or("");
+ if let Some(tag) = Self::modifier_to_tag(key, value) {
+ tags.push(tag);
+ }
+ }
+
+ if tags.is_empty() {
+ String::new()
+ } else {
+ format!("{} ", tags.join(" "))
+ }
+ }
+}
+
+impl Formatter for ElevenLabsV3Formatter {
+ fn format(&self, ast: &AstNode) -> Result {
+ let mut out = String::new();
+ self.format_node_internal(ast, &mut out)?;
+ Ok(out)
+ }
+
+ fn format_node(&self, node: &AstNode) -> Result {
+ let mut out = String::new();
+ self.format_node_internal(node, &mut out)?;
+ Ok(out)
+ }
+}
+
+#[cfg(test)]
+mod tests {
+ use crate::formatters::base::Platform;
+ use crate::parser::SpeechMarkdownParser;
+
+ fn to_v3(input: &str) -> String {
+ SpeechMarkdownParser::to_ssml(input, Platform::ElevenLabsV3).unwrap()
+ }
+
+ #[test]
+ fn short_breaks_map_to_pause_steps() {
+ assert_eq!(to_v3("Sample [250ms] speech"), "Sample ... speech");
+ assert_eq!(to_v3("Sample [0.5s] speech"), "Sample [pause] speech");
+ assert_eq!(to_v3("Sample [2s] speech"), "Sample [long pause] speech");
+ }
+
+ #[test]
+ fn break_strengths_map_to_pause_tags() {
+ assert_eq!(to_v3("[break:\"none\"]"), "");
+ assert_eq!(to_v3("[break:\"x-weak\"]"), "...");
+ assert_eq!(to_v3("[break:\"weak\"]"), "[short pause]");
+ assert_eq!(to_v3("[break:\"medium\"]"), "[pause]");
+ assert_eq!(to_v3("[break:\"strong\"]"), "[long pause]");
+ assert_eq!(to_v3("[break:\"x-strong\"]"), "[long pause]");
+ assert_eq!(to_v3("[break:\"bogus\"]"), "[pause]");
+ }
+
+ #[test]
+ fn emphasis_maps_to_tags_not_caps() {
+ assert_eq!(to_v3("very ++important++"), "very [emphasized] important");
+ assert_eq!(
+ to_v3("a +little+ bit"),
+ "a [stress on next word] little bit"
+ );
+ assert_eq!(to_v3("a -little- bit"), "a [understated] little bit");
+ assert_eq!(to_v3("~whatever~"), "whatever");
+ }
+
+ #[test]
+ fn whisper_and_emotion_modifiers_prefix_tags() {
+ assert_eq!(
+ to_v3("(it's a secret)[whisper]"),
+ "[whispers] it's a secret"
+ );
+ assert_eq!(to_v3("(great news)[excited]"), "[excited] great news");
+ }
+
+ #[test]
+ fn rate_and_volume_map_to_tempo_tags() {
+ assert_eq!(
+ to_v3("(read this)[rate:\"fast\";volume:\"loud\"]"),
+ "[rushed] [loudly] read this"
+ );
+ assert_eq!(to_v3("(slowly)[rate:\"slow\"]"), "[drawn out] slowly");
+ }
+
+ #[test]
+ fn unsupported_modifiers_degrade_to_text() {
+ assert_eq!(to_v3("(hello)[voice:\"Brian\"]"), "hello");
+ assert_eq!(to_v3("(bonjour)[lang:\"fr-FR\"]"), "bonjour");
+ assert_eq!(to_v3("(42)[number]"), "42");
+ }
+
+ #[test]
+ fn ipa_becomes_native_slash_form() {
+ assert_eq!(to_v3("(speech)/spitʃ/"), "\"/spitʃ/\"");
+ assert_eq!(to_v3("(word)[ipa:\"wɜːd\"]"), "\"/wɜːd/\"");
+ }
+
+ #[test]
+ fn sub_speaks_alias() {
+ assert_eq!(to_v3("{AL}aluminum"), "aluminum");
+ }
+
+ #[test]
+ fn sections_become_prefix_tags() {
+ assert_eq!(to_v3("#[excited] Hello world"), "[excited] Hello world");
+ // Unknown styles pass through as natural-language direction.
+ assert_eq!(to_v3("#[sarcastic] nice"), "[sarcastic] nice");
+ // Recognized section modifiers map like inline ones.
+ assert_eq!(to_v3("#[rate:\"slow\"] steady"), "[drawn out] steady");
+ // defaults produces nothing.
+ assert_eq!(to_v3("#[defaults] plain"), " plain");
+ }
+
+ #[test]
+ fn expressive_tags_pass_through() {
+ assert_eq!(
+ to_v3("He [laugh] and then [applause] left"),
+ "He [laugh] and then [applause] left"
+ );
+ }
+
+ #[test]
+ fn audio_and_mark_are_dropped() {
+ // Dropped nodes leave the surrounding spacing in place.
+ assert_eq!(
+ to_v3("Hello [mark:chapter1]  world"),
+ "Hello world"
+ );
+ }
+
+ #[test]
+ fn no_speak_wrapper_and_no_escaping() {
+ assert_eq!(to_v3("1 < 2 & 3 > 0"), "1 < 2 & 3 > 0");
+ }
+}
diff --git a/src/formatters/mod.rs b/src/formatters/mod.rs
index 2017083..e35ec69 100644
--- a/src/formatters/mod.rs
+++ b/src/formatters/mod.rs
@@ -1,4 +1,6 @@
pub mod base;
+pub mod elevenlabs;
+pub mod elevenlabs_v3;
pub mod ssml;
pub mod text;
diff --git a/src/formatters/ssml/base.rs b/src/formatters/ssml/base.rs
index 5fd4496..ec8711a 100644
--- a/src/formatters/ssml/base.rs
+++ b/src/formatters/ssml/base.rs
@@ -232,6 +232,7 @@ impl SsmlFormatterBase {
NodeType::ShortIpa => self.format_ipa(node),
NodeType::BareIpa => self.format_bare_ipa(node),
NodeType::ShortSub => self.format_short_sub(node),
+ NodeType::Expressive => Ok(format!("[{}]", node.text)),
_ => Ok(node.text.clone()),
}
}
@@ -855,12 +856,7 @@ mod phonetic_alphabet_tests {
let (tag, attrs) = fmt().attribute_to_tag(key, src).unwrap();
assert_eq!(tag, "phoneme", "key {}", key);
assert_eq!(attrs_get(&attrs, "alphabet"), Some("ipa"), "key {}", key);
- assert_eq!(
- attrs_get(&attrs, "ph"),
- Some(expected_ipa),
- "key {}",
- key
- );
+ assert_eq!(attrs_get(&attrs, "ph"), Some(expected_ipa), "key {}", key);
}
}
diff --git a/src/formatters/text.rs b/src/formatters/text.rs
index 6a525a5..ed27d5f 100644
--- a/src/formatters/text.rs
+++ b/src/formatters/text.rs
@@ -113,7 +113,12 @@ impl TextFormatter {
// Mark tags - no output
NodeType::Mark => {
- // Mark tags produce no output
+ // Mark tags produce no text output
+ }
+
+ // Expressive audio tags - kept verbatim ([laugh], [sigh], …)
+ NodeType::Expressive => {
+ result.push(format!("[{}]", node.text));
}
// Modifiers are handled as part of text modifiers
diff --git a/src/parser/parser.rs b/src/parser/parser.rs
index 4e05359..1e85678 100644
--- a/src/parser/parser.rs
+++ b/src/parser/parser.rs
@@ -7,6 +7,66 @@ use crate::ssml_to_smd;
pub struct SpeechMarkdownParser;
+/// Expressive audio tags recognized by the grammar, mirroring the
+/// speechmarkdown-js reference implementation. Eleven v3 treats any
+/// bracketed natural-language cue as an audio tag; this list covers the
+/// fixed vocabulary the JS parser formalizes. Anything else bracketed
+/// falls through as plain text (and therefore also passes through to
+/// Eleven v3 verbatim).
+const EXPRESSIVE_TAGS: &[&str] = &[
+ "laugh",
+ "laughter",
+ "sigh",
+ "cough",
+ "cheer",
+ "cheering",
+ "cry",
+ "crying",
+ "gasp",
+ "groan",
+ "groaning",
+ "hum",
+ "hmm",
+ "mm-hmm",
+ "oh",
+ "sniff",
+ "whew",
+ "wow",
+ "yawn",
+ "yeah",
+ "huh",
+ "tsk",
+ "uh-huh",
+ "mmm",
+ "mhm",
+ "ahem",
+ "applause",
+ "boo",
+ "giggle",
+ "hiccup",
+ "hurray",
+ "moan",
+ "pant",
+ "scream",
+ "shush",
+ "sneeze",
+ "throat-clear",
+ "wheeze",
+ "whimper",
+ "yay",
+ "bleh",
+ "eek",
+ "meh",
+ "ooh",
+ "pfft",
+ "phew",
+ "psst",
+ "shh",
+ "tsk-tsk",
+ "uh-oh",
+ "umph",
+];
+
impl SpeechMarkdownParser {
/// Parse SpeechMarkdown text into an AST
pub fn parse(input: &str) -> Result {
@@ -124,6 +184,9 @@ impl SpeechMarkdownParser {
NodeType::ShortBreak,
format!("[{}]", bracket_content),
));
+ } else if Self::is_expressive_tag(&bracket_content) {
+ document = document
+ .add_child(AstNode::new(NodeType::Expressive, bracket_content));
} else {
current_text.push('[');
current_text.push_str(&bracket_content);
@@ -497,7 +560,31 @@ impl SpeechMarkdownParser {
}
fn is_time_break(s: &str) -> bool {
- s.ends_with("s") || s.ends_with("ms")
+ let Some(body) = s.strip_suffix("ms").or_else(|| s.strip_suffix('s')) else {
+ return false;
+ };
+ // Strict numeric body matching the speechmarkdown-js grammar
+ // (\d+(\.\d+)?): digits, with an optional decimal point that must
+ // be surrounded by digits. Rejects words ending in 's' ("apps"),
+ // malformed numbers ("1.2.3s", "1..5s"), sign/exponent spellings
+ // ("+2s", "1e3s"), and bare/edge dots (".5s", "5.s", "0.s").
+ let mut parts = body.split('.');
+ match (parts.next(), parts.next(), parts.next()) {
+ (Some(int_part), None, None) => {
+ !int_part.is_empty() && int_part.bytes().all(|b| b.is_ascii_digit())
+ }
+ (Some(int_part), Some(frac_part), None) => {
+ !int_part.is_empty()
+ && int_part.bytes().all(|b| b.is_ascii_digit())
+ && !frac_part.is_empty()
+ && frac_part.bytes().all(|b| b.is_ascii_digit())
+ }
+ _ => false,
+ }
+ }
+
+ fn is_expressive_tag(s: &str) -> bool {
+ EXPRESSIVE_TAGS.contains(&s)
}
fn read_until(chars: &mut std::iter::Peekable, end: char) -> (String, bool) {
@@ -578,11 +665,43 @@ mod tests {
println!("==========================");
}
+ #[test]
+ fn test_parse_expressive_tag() {
+ let ast = SpeechMarkdownParser::parse("He [laugh] and then [applause] left").unwrap();
+ let tags: Vec<&AstNode> = ast
+ .children
+ .iter()
+ .filter(|c| c.node_type == NodeType::Expressive)
+ .collect();
+ assert_eq!(tags.len(), 2);
+ assert_eq!(tags[0].text, "laugh");
+ assert_eq!(tags[1].text, "applause");
+ }
+
+ #[test]
+ fn test_parse_expressive_untouched_when_not_keyword() {
+ // Unknown bracketed content stays plain text (passthrough for
+ // Eleven v3's open-ended natural-language audio tags).
+ let ast = SpeechMarkdownParser::parse("This [not-a-known-tag] stays").unwrap();
+ assert!(ast
+ .children
+ .iter()
+ .all(|c| c.node_type == NodeType::PlainText));
+ }
+
+ #[test]
+ fn test_expressive_text_output_keeps_tag() {
+ let out = SpeechMarkdownParser::to_text("He said [boo] and left").unwrap();
+ assert_eq!(out, "He said [boo] and left");
+ }
+
#[test]
fn test_is_speech_markdown() {
assert!(!SpeechMarkdownParser::is_speech_markdown("Hello world"));
assert!(!SpeechMarkdownParser::is_speech_markdown(""));
- assert!(SpeechMarkdownParser::is_speech_markdown("Hello (world)[emphasis:\"strong\"]"));
+ assert!(SpeechMarkdownParser::is_speech_markdown(
+ "Hello (world)[emphasis:\"strong\"]"
+ ));
assert!(SpeechMarkdownParser::is_speech_markdown("Sample [2s] text"));
assert!(SpeechMarkdownParser::is_speech_markdown("++strong++"));
assert!(SpeechMarkdownParser::is_speech_markdown("~word~"));
diff --git a/tests/integration_test.rs b/tests/integration_test.rs
index 414a3c4..353b799 100644
--- a/tests/integration_test.rs
+++ b/tests/integration_test.rs
@@ -116,6 +116,23 @@ fn test_all_test_cases() {
}
}
+ // Test prompt markup output for ElevenLabs (pre-v3 dialect)
+ let elevenlabs_file = test_dir.join(format!("{}.elevenlabs.ssml", test_name));
+ if elevenlabs_file.exists() {
+ let expected = fs::read_to_string(&elevenlabs_file).unwrap_or_else(|_| {
+ panic!("Failed to read ElevenLabs file: {:?}", elevenlabs_file)
+ });
+
+ let result = SpeechMarkdownParser::to_ssml(&input, Platform::ElevenLabs);
+ if result.is_err() {
+ all_checks_passed = false;
+ } else if let Ok(actual) = result {
+ if actual.trim() != normalize_line_endings(expected.trim()) {
+ all_checks_passed = false;
+ }
+ }
+ }
+
all_checks_passed
}
Err(_e) => false,
@@ -143,14 +160,16 @@ fn test_all_test_cases() {
for test in &failed_tests {
println!(" - {}", test);
}
- }
-
- // Only panic if we have significant failures (>50% fail rate)
- if failed > 0 && (failed as f64 / (passed + failed) as f64) > 0.5 {
+ // Zero tolerance: this suite is the regression net for the
+ // enforced fixture families (.txt, .alexa.ssml, .google.ssml,
+ // .elevenlabs.ssml). Any failure should fail CI — a lenient
+ // threshold would let a broken formatter slip through while
+ // enough unrelated cases still pass.
panic!(
- "Too many test failures: {}/{} failed",
+ "{}/{} corpus tests failed: {:?}",
failed,
- passed + failed
+ passed + failed,
+ failed_tests
);
}
}