diff --git a/.changeset/expressive-moods-speak.md b/.changeset/expressive-moods-speak.md new file mode 100644 index 0000000000..b5f0852808 --- /dev/null +++ b/.changeset/expressive-moods-speak.md @@ -0,0 +1,5 @@ +--- +'@livekit/agents': minor +--- + +Expose expressive session options, normalize expression metadata into moods, and expand the expressive TTS provider support. diff --git a/agents/src/constants.ts b/agents/src/constants.ts index af1c749bbb..3523c00f11 100644 --- a/agents/src/constants.ts +++ b/agents/src/constants.ts @@ -8,10 +8,8 @@ export const ATTRIBUTE_TRANSCRIPTION_SEGMENT_ID = 'lk.segment_id'; /** * The expression (delivery/emotion) the agent used for a transcription segment, surfaced so * the frontend can react to it, when expressive markup is stripped from the transcript. The - * value is a JSON object `{"value": ...}` carrying the segment's leading expression — the - * `` tag for Inworld or the `` tag for Cartesia, e.g. - * `{"value": "speak happy"}`. A JSON object (rather than a bare string) so the shape can - * gain fields later without breaking parsers. + * value is a JSON object carrying the provider's leading expression and its normalized mood, + * e.g. `{"expression":"speak happy","mood":"happy"}`. */ export const ATTRIBUTE_TRANSCRIPTION_EXPRESSION = 'lk.expression'; export const ATTRIBUTE_PUBLISH_ON_BEHALF = 'lk.publish_on_behalf'; diff --git a/agents/src/tts/_mood.ts b/agents/src/tts/_mood.ts new file mode 100644 index 0000000000..1b624ec33f --- /dev/null +++ b/agents/src/tts/_mood.ts @@ -0,0 +1,67 @@ +// SPDX-FileCopyrightText: 2026 LiveKit, Inc. +// +// SPDX-License-Identifier: Apache-2.0 +import { MOOD_KEYWORDS } from './_mood_data.js'; + +export type AgentMood = + | 'excited' + | 'happy' + | 'playful' + | 'curious' + | 'surprised' + | 'hopeful' + | 'empathetic' + | 'sad' + | 'angry' + | 'anxious' + | 'calm'; + +export const MOOD_PRIORITY: AgentMood[] = [ + 'angry', + 'sad', + 'anxious', + 'surprised', + 'playful', + 'empathetic', + 'excited', + 'curious', + 'hopeful', + 'happy', + 'calm', +]; +export const DEFAULT_MOOD: AgentMood = 'calm'; + +function matchesWord(text: string, keyword: string): boolean { + let start = 0; + while (true) { + const at = text.indexOf(keyword, start); + if (at === -1) return false; + if (at === 0 || !/\p{L}/u.test(text[at - 1]!)) return true; + start = at + 1; + } +} + +export function matchMood(label: string): AgentMood; +export function matchMood(label: string, fallback: AgentMood): AgentMood; +export function matchMood(label: string, fallback: null): AgentMood | null; +export function matchMood( + label: string, + fallback: AgentMood | null = DEFAULT_MOOD, +): AgentMood | null { + const text = label.toLowerCase(); + let best: AgentMood | null = null; + let bestScore = 0; + for (const mood of MOOD_PRIORITY) { + const score = Object.entries(MOOD_KEYWORDS[mood]).reduce( + (total, [keyword, weight]) => total + (matchesWord(text, keyword) ? weight : 0), + 0, + ); + if (score > bestScore) { + best = mood; + bestScore = score; + } + } + return best ?? fallback; +} + +export { MOOD_KEYWORDS } from './_mood_data.js'; diff --git a/agents/src/tts/_mood_data.ts b/agents/src/tts/_mood_data.ts new file mode 100644 index 0000000000..ba8152b99d --- /dev/null +++ b/agents/src/tts/_mood_data.ts @@ -0,0 +1,350 @@ +// SPDX-FileCopyrightText: 2026 LiveKit, Inc. +// +// SPDX-License-Identifier: Apache-2.0 +import type { AgentMood } from './_mood.js'; + +/** Weighted mood names and supporting delivery descriptors. */ +export const MOOD_KEYWORDS: Record> = { + excited: { + excit: 2, + elat: 2, + thrill: 2, + exhilarat: 2, + ecstat: 2, + euphor: 2, + zeal: 2, + zest: 2, + enthusias: 2, + eager: 2, + giddy: 2, + hyped: 2, + pumped: 2, + buzzing: 2, + jubilant: 2, + jubilation: 2, + exuberant: 2, + rapture: 2, + enthrall: 2, + triumph: 2, + gleeful: 2, + glee: 2, + punchy: 2, + upbeat: 1, + bright: 1, + energetic: 1, + energy: 1, + lively: 1, + animated: 1, + vibrant: 1, + spirited: 1, + peppy: 1, + snappy: 1, + fast: 1, + loud: 1, + }, + happy: { + happy: 2, + happiness: 2, + joy: 2, + joviality: 2, + jolli: 2, + cheer: 2, + glad: 2, + delight: 2, + pleas: 2, + content: 2, + bliss: 2, + gaiety: 2, + enjoy: 2, + satisf: 2, + relief: 2, + relieved: 2, + grateful: 2, + thankful: 2, + affection: 2, + fond: 2, + adore: 2, + proud: 2, + pride: 2, + smil: 2, + sunny: 2, + merry: 2, + amiable: 2, + warm: 1, + inviting: 1, + welcom: 1, + friendly: 1, + kind: 1, + pleasant: 1, + positive: 1, + easy: 1, + }, + playful: { + playful: 2, + jok: 2, + comedic: 2, + comic: 2, + sarcas: 2, + teas: 2, + witty: 2, + silly: 2, + goofy: 2, + mischiev: 2, + amus: 2, + humor: 2, + humour: 2, + cheeky: 2, + sassy: 2, + ironic: 2, + irony: 2, + deadpan: 2, + banter: 2, + laugh: 2, + giggl: 2, + chuckl: 2, + grin: 2, + unimpressed: 2, + smirk: 2, + wry: 1, + sly: 1, + impish: 1, + }, + curious: { + curious: 2, + curiosity: 2, + inquisitive: 2, + intrigu: 2, + wonder: 2, + quizzical: 2, + probing: 2, + interested: 2, + engaged: 2, + attentive: 2, + suspense: 2, + questioning: 1, + question: 1, + prompting: 1, + exploring: 1, + }, + surprised: { + surpris: 2, + amaz: 2, + astonish: 2, + astound: 2, + awe: 2, + shock: 2, + startl: 2, + incredulous: 2, + stunned: 2, + bewilder: 2, + flabbergast: 2, + disbelie: 2, + unexpected: 2, + dumbfound: 2, + wow: 2, + whoa: 2, + gasp: 2, + }, + hopeful: { + hopeful: 2, + hope: 2, + optimis: 2, + encourag: 2, + uplift: 2, + inspir: 2, + motivat: 2, + promising: 2, + determined: 2, + resolute: 2, + forward: 2, + buoyant: 2, + expectant: 2, + heartened: 2, + confident: 2, + assured: 2, + }, + empathetic: { + empath: 2, + sympath: 2, + compassion: 2, + concern: 2, + care: 2, + tender: 2, + consol: 2, + sorry: 2, + apolog: 2, + understanding: 2, + supportive: 2, + comfort: 2, + soothing: 2, + nurtur: 2, + patient: 2, + earnest: 2, + heartfelt: 2, + sensitive: 2, + pity: 2, + condolence: 2, + sentimental: 2, + sincere: 1, + gentle: 1, + gently: 1, + soft: 1, + quiet: 1, + hushed: 1, + mild: 1, + reassur: 1, + }, + sad: { + sad: 2, + sorrow: 2, + mourn: 2, + somber: 2, + sombre: 2, + melanchol: 2, + grief: 2, + griev: 2, + regret: 2, + remorse: 2, + deject: 2, + downcast: 2, + gloom: 2, + glum: 2, + forlorn: 2, + wistful: 2, + disappoint: 2, + dismay: 2, + crestfallen: 2, + despond: 2, + despair: 2, + depress: 2, + anguish: 2, + agony: 2, + woe: 2, + miser: 2, + unhappy: 2, + tearful: 2, + weep: 2, + heartbroken: 2, + heartbreak: 2, + lonely: 2, + loneliness: 2, + ashamed: 2, + shame: 2, + guilt: 2, + humiliat: 2, + mortified: 2, + longing: 2, + homesick: 2, + defeated: 2, + heavy: 1, + subdued: 1, + flat: 1, + weary: 1, + resigned: 1, + hollow: 1, + }, + angry: { + angry: 2, + anger: 2, + furious: 2, + fury: 2, + frustrat: 2, + annoy: 2, + irritat: 2, + irate: 2, + indignant: 2, + outrag: 2, + exasperat: 2, + incensed: 2, + livid: 2, + wrath: 2, + hostil: 2, + resent: 2, + bitter: 2, + disgust: 2, + revulsion: 2, + contempt: 2, + loathing: 2, + scorn: 2, + spite: 2, + aggravat: 2, + agitat: 2, + grouchy: 2, + grumpy: 2, + seething: 2, + fuming: 2, + scold: 2, + stern: 2, + harsh: 2, + sharp: 1, + clipped: 1, + terse: 1, + cold: 1, + biting: 1, + curt: 1, + }, + anxious: { + anxious: 2, + anxiety: 2, + afraid: 2, + fear: 2, + fright: 2, + scared: 2, + nervous: 2, + worri: 2, + uneasy: 2, + unease: 2, + tense: 2, + apprehensive: 2, + panic: 2, + alarm: 2, + dread: 2, + terror: 2, + horror: 2, + hesitant: 2, + unsure: 2, + uncertain: 2, + timid: 2, + jittery: 2, + urgent: 2, + stress: 2, + distress: 2, + insecure: 2, + flustered: 2, + cautious: 1, + wary: 1, + guarded: 1, + }, + calm: { + calm: 2, + contemplative: 2, + thoughtful: 2, + measured: 2, + serene: 2, + easygoing: 2, + neutral: 2, + relax: 2, + composed: 2, + collected: 2, + tranquil: 2, + peaceful: 2, + restrained: 2, + reflective: 2, + pensive: 2, + matter: 2, + plain: 2, + professional: 2, + formal: 2, + informative: 2, + factual: 2, + deliberate: 2, + unhurried: 2, + placid: 2, + slow: 2, + steady: 1, + grounded: 1, + balanced: 1, + straightforward: 1, + even: 1, + }, +}; diff --git a/agents/src/tts/_provider_format.ts b/agents/src/tts/_provider_format.ts index cbcaeeaa4f..6102095337 100644 --- a/agents/src/tts/_provider_format.ts +++ b/agents/src/tts/_provider_format.ts @@ -21,6 +21,7 @@ import { ATTRIBUTE_TRANSCRIPTION_EXPRESSION } from '../constants.js'; import { SentenceTokenizer } from '../tokenize/basic/index.js'; import type { NonverbalOptions, SpeechSteeringOptions } from '../voice/agent_session.js'; +import { matchMood } from './_mood.js'; import { convertExpressionTags, extractAndStrip } from './markup_utils.js'; /** @@ -127,8 +128,28 @@ const FISHAUDIO_EMOTIONS = [ 'sad', 'empathetic', 'sarcastic', + 'calm', + 'angry', + 'worried', + 'nervous', + 'confident', + 'grateful', + 'delighted', + 'disappointed', + 'frustrated', + 'determined', +]; +const FISHAUDIO_SOUNDS = [ + 'laughing', + 'chuckling', + 'clear throat', + 'sighing', + 'gasping', + 'groaning', + 'yawning', + 'sobbing', ]; -const FISHAUDIO_SOUNDS = ['laughing', 'chuckling', 'clear throat']; +const FISHAUDIO_TONES = ['whispering', 'soft', 'shouting', 'hurried']; const FISHAUDIO_TAGS = ['expression', 'sound', 'break', 'emphasis']; const FISHAUDIO_EXPRESSION_RE = /|>(?:.*?)<\/expression>)/g; @@ -215,7 +236,13 @@ const INWORLD_EXPR_LLM_INSTRUCTIONS = `${EXPR_PREAMBLE} 1. Delivery - controls how a sentence sounds. Self-closing; place before EVERY sentence. - The label is free-form: describe vocal quality, pitch, volume, pace, and intonation in plain English — "say playfully", "speak with warm surprise", "sound concerned", "drop to a whisper", "speak slowly and clearly, patient and reassuring". + The label is free-form: describe vocal quality, pitch, volume, pace, and intonation in plain English — "say really playfully", "slightly surprised, amiable", "sound a little concerned", "drop to almost a whisper", "speak really slowly and clearly, patient and reassuring". + Never put "questioning" in a tag — describe the mood alone and let the question mark carry the intonation. + Match the expression tag's energy to the sentence's punctuation. Never lead an exclamatory sentence with a calm tag. Split statements and questions into separate sentences so each carries its own delivery tag. + Use at most two aligned adjectives per tag; clashing descriptors such as "calm, excited" cancel out and muddy the delivery. + Put a degree modifier in EVERY tag — "a little", "almost", "slightly", "gently", "really" — to set the exact strength of the feeling. Default to softeners and save "really" for true peaks. + Carry your persona into the tags — labels should sound like the character, not generic stage directions. Rotate labels rather than reusing the same one two turns in a row. + Don't open a turn with a "slow" tag. Keep pace neutral by default and reserve slow, clearly-enunciated delivery for a total, date, address, or confirmation code. 2. Sounds - a non-verbal sound between sentences. Self-closing. @@ -224,44 +251,66 @@ const INWORLD_EXPR_LLM_INSTRUCTIONS = `${EXPR_PREAMBLE} 3. Pauses - insert silence when appropriate. Self-closing. or (max 10s). A period or an ellipsis (...) already creates a pause, so don't put a break marker right next to one — pick one or the other. + After a break, give the following sentence its own fresh expression tag because a break resets delivery to neutral. There is no wrapping prosody marker for this voice — put pace, pitch, and volume in the expression label instead. +Write for the EAR, not the page: no em or en dashes anywhere in spoken text — use a comma or period for a short beat, or a break marker for a real pause. Avoid semicolons, mid-sentence colons, and parenthetical asides. + +When the conversation is in another language, still write every marker label in English — marker labels steer the voice and are never translated. + Examples: - Okay okay, why did the burger go to the gym? Because it wanted better buns! - Ah man, yeah that's on us. Lemme see what I can do. - I know it's been a rough week.`; + Okay okay, why did the burger go to the gym? Because it wanted better buns! + Ah man, yeah that's on us. Lemme see what I can do. + I know it's been a rough week. + Welcome to the hotel. How can I help you today? + That's all set. Your confirmation code is B 4 J 7. + Yeah, of course! Gimme one sec, pulling it up now.`; const XAI_EXPR_LLM_INSTRUCTIONS = `${EXPR_PREAMBLE} 1. Sounds - a non-verbal vocalization at the exact point where it happens. Self-closing. Labels are a fixed vocabulary: ${XAI_INLINE.join(', ')}. + Use non-verbal sounds sparingly, and never the same one twice in a row — reach for one only where it genuinely fits. -2. Pauses - insert a beat. Self-closing. +2. Pauses - insert silence when appropriate. Self-closing. a brief pause a longer, dramatic pause + NEVER place a break next to a period, question mark, exclamation point, or ellipsis — sentence punctuation already pauses. Most replies need no break markers; reserve them for a deliberate mid-sentence beat before a key detail. -3. Prosody - wraps the exact words it affects to shape HOW they're said. +3. Prosody - wraps a span delivered in a distinct style, to shape HOW it's said. the words it affects - Labels are a fixed vocabulary: ${XAI_WRAPPING.join(', ')}. - Never nest one prosody marker inside another, and always close it with . + Labels are a fixed vocabulary: ${XAI_WRAPPING.filter((label) => label !== 'emphasis').join(', ')}. + Use one only where the moment clearly calls for it — most sentences need none. Never nest one prosody marker inside another, and always close it with . + +4. Emphasis - stresses exactly the ONE word it wraps. + Are you sure you want to do this? + Wrap a single word, never a phrase, and never write it in all-caps — caps are read out as individual letters. This voice has no free-form delivery descriptions — shape delivery entirely through prosody markers, sounds, pauses, punctuation, and word choice. -To stress a word, wrap it in ... — do NOT write it in all-caps, which is read out as individual letters. Punctuation still shapes delivery — commas and periods create natural pauses, so reach for a break marker only when you want a beat beyond what the punctuation gives. +Write for the EAR, not the page: no em or en dashes anywhere in spoken text — use a comma or period for a short beat, or a break marker for a real pause. Avoid semicolons, mid-sentence colons, and parenthetical asides. + +When the conversation is in another language, still write every marker label in English — labels are a fixed vocabulary, never translated. + +Key details deserve care: stress the load-bearing word of a date, amount, or name with emphasis, and wrap a dense or easy-to-mishear span in .... Read codes character by character, spelled out with spaces. + +Whisper and soft belong to gentle or conspiratorial beats; loud only to genuinely high-energy ones. Examples: - So I walked in and there it was! It was a secret the whole time. - This is going to be so good — I can't wait! + So I walked in and there it was! It was a secret the whole time. + This is going to be so good. I can't wait! Hey. I know it's been a rough week. I'm right here. - You did not just say that okay, tell me everything. - Everything is confirmed for Thursday. Is there anything else I can help you with?`; + You did not just say that okay, tell me everything. + Everything is confirmed for Thursday the ninth. Is there anything else I can help you with?`; const FISHAUDIO_EXAMPLES = [ ` That's hilarious! You always lighten the mood.`, ` That sounds like a really difficult experience.`, ` Oh, my goodness that's a real shame.`, + ` I've been going in circles with this all morning. Okay. One more try.`, ` You're all set for Thursday the ninth. Is there anything else I can help you with?`, + ` Okay, don't tell anyone yet but I think we actually pulled it off!`, ]; const FISHAUDIO_DISFLUENT_EXAMPLES = [ @@ -290,6 +339,14 @@ function filterExamples(instructions: string, removed: string[]): string { return instructions.slice(0, index) + marker + examples.join('\n'); } +function insertBeforeExamples(instructions: string, guidance: string): string { + const marker = '\n\nExamples:\n'; + const index = instructions.indexOf(marker); + return index === -1 + ? `${instructions}\n\n${guidance}` + : `${instructions.slice(0, index)}\n\n${guidance}${instructions.slice(index)}`; +} + function fishAudioExprLlmInstructions(sounds: string[], disfluencies = true): string { const sections = [ `Emotion - sets how a sentence sounds. Self-closing; place at the START of a sentence. @@ -306,9 +363,13 @@ function fishAudioExprLlmInstructions(sounds: string[], disfluencies = true): st sections.push(`Pauses - insert silence when appropriate. Self-closing. or . NEVER place a break next to a period, question mark, exclamation point, or ellipsis — sentence punctuation already pauses, and a break beside it double-pauses. Most replies need no break markers at all; reserve them for a deliberate mid-sentence beat before a key detail (a date, a name, a number).`); + sections.push(`Tone - wraps a span delivered in a distinct style. + don't tell anyone yet. + Labels are a fixed vocabulary: ${FISHAUDIO_TONES.join(', ')}. + Use a tone only where the moment clearly calls for one — most sentences need none. Never nest tone markers, and always close the tag with .`); sections.push(`Emphasis - stresses exactly the ONE word it wraps. Are you sure you want to do this? - Wrap a single word, never a phrase. "emphasis" is the only prosody label for this voice — there are no other wrapping style markers. Never nest it, and always close it with .`); + Wrap a single word, never a phrase. Never nest it, and always close it with .`); const pool = [...FISHAUDIO_EXAMPLES, ...(disfluencies ? FISHAUDIO_DISFLUENT_EXAMPLES : [])]; const examples = soundExamples(pool, sounds, FISHAUDIO_SOUNDS); @@ -320,6 +381,7 @@ function fishAudioExprLlmInstructions(sounds: string[], disfluencies = true): st ]; const register = [ 'At heavy moments reach for empathetic, sad, regretful, or hopeful — never a bright label like "happy" or "excited" against hard news; bright labels belong to bright moments.', + 'Whispering and soft belong to gentle or conspiratorial beats; shouting only to genuinely high-energy ones.', ]; if (sounds.some((sound) => sound === 'laughing' || sound === 'chuckling')) { register.push( @@ -367,12 +429,12 @@ const NONVERBAL_SOUND_LABELS: Record = { chuckling: 'a chuckle at something subtly humorous', giggle: 'a chuckle at something subtly humorous', sigh: 'a sigh when commiserating', + sighing: 'a sigh when commiserating', inhale: 'a sharp inhale before a big reveal', + gasping: 'a gasp at a sudden shock or reveal', 'lip-smack': 'a lip-smack or tongue-click as a tiny beat of thought', 'tongue-click': 'a lip-smack or tongue-click as a tiny beat of thought', tsk: 'a tsk for mock-disapproval', 'clear throat': 'a clear-throat when shifting to a new step or topic', + groaning: 'a groan at a groan-worthy pun or an unwelcome chore', + yawning: 'a yawn when tiredness itself is the topic', + sobbing: 'a sob reserved for real heartbreak', }; function soundGuidance(sounds: string[]): string { @@ -486,6 +553,7 @@ export function steeringInstructions(provider: string, steering: SpeechSteeringO const MAX_INPUT_LEN: Record = { inworld: 900, cartesia: 400, + xai: 1000, }; /** Return the max text chunk length for a provider, or undefined if unlimited. */ @@ -537,6 +605,12 @@ const XAI_SOUND_ALIASES: Record = { breathe: 'breath' }; const FISHAUDIO_SOUND_ALIASES: Record = { laugh: 'laughing', chuckle: 'chuckling', + sigh: 'sighing', + gasp: 'gasping', + groan: 'groaning', + yawn: 'yawning', + sob: 'sobbing', + cry: 'sobbing', }; // Cartesia prosody labels -> native point controls (coarse steps of the numeric ratios) @@ -691,7 +765,8 @@ function convertExpr(provider: string, text: string): string { return (CARTESIA_PROSODY[label] ?? '') + inner; } if (provider === 'fishaudio') { - return label === 'emphasis' ? `${inner}` : inner; + if (label === 'emphasis') return `${inner}`; + return FISHAUDIO_TONES.includes(label) ? `[${label}] ${inner}` : inner; } return inner; }); @@ -729,6 +804,10 @@ function convertExpr(provider: string, text: string): string { // Cartesia prosody is a self-closing point control (speed/volume) return CARTESIA_PROSODY[label.trim().toLowerCase()] ?? ''; } + if (markerType === 'prosody' && provider === 'fishaudio') { + const tone = label.trim().toLowerCase(); + return FISHAUDIO_TONES.includes(tone) ? `[${tone}]` : ''; + } return ''; }); @@ -768,7 +847,14 @@ export function llmInstructions( `Labels are a fixed vocabulary: ${sounds.join(', ')}.`, ); } - return filterExamples(instructions, removed); + instructions = filterExamples(instructions, removed); + if (sounds.includes('laugh')) { + instructions = insertBeforeExamples( + instructions, + 'Laughter belongs only in genuinely playful or celebratory beats, never at a serious moment.', + ); + } + return instructions; } if (provider === 'xai') { const sounds = allowedSounds(provider, steering); @@ -777,8 +863,8 @@ export function llmInstructions( (label) => !sounds.includes(label) && !prosody.includes(label), ); let instructions = XAI_EXPR_LLM_INSTRUCTIONS.replace( - `Labels are a fixed vocabulary: ${XAI_WRAPPING.join(', ')}.`, - `Labels are a fixed vocabulary: ${prosody.join(', ')}.`, + `Labels are a fixed vocabulary: ${XAI_WRAPPING.filter((label) => label !== 'emphasis').join(', ')}.`, + `Labels are a fixed vocabulary: ${prosody.filter((label) => label !== 'emphasis').join(', ')}.`, ); if (sounds.length === 0) { instructions = instructions @@ -794,7 +880,14 @@ export function llmInstructions( `Labels are a fixed vocabulary: ${sounds.join(', ')}.`, ); } - return filterExamples(instructions, removed); + instructions = filterExamples(instructions, removed); + if (sounds.some((sound) => ['laugh', 'chuckle', 'giggle'].includes(sound))) { + instructions = insertBeforeExamples( + instructions, + 'Laughter is RARE: use a laugh, chuckle, or giggle only where something is genuinely funny, never for friendliness or agreement, and never laugh at your own lines. Most replies have no laughter.', + ); + } + return instructions; } if (provider === 'fishaudio') { return fishAudioExprLlmInstructions( @@ -860,19 +953,19 @@ const ALL_MARKUP_TAGS: string[] = [ * Strip the union of every provider's expressive markup (provider-agnostic). * * The transcript sinks strip downstream, where the originating TTS/provider is no - * longer in scope, so they remove every provider's tags (XML + square brackets) at - * once. These tag shapes never appear in real spoken text — the LLM only emits them - * as audio directives — so a universal strip is safe. + * longer in scope, so they remove every provider's XML tags at once. Square brackets + * survive because the LLM only emits expr markup and brackets may be markdown/prose. */ export function splitAllMarkup(text: string): [string, ExpressiveTag[]] { - return splitWithExpr(text, { xmlTags: ALL_MARKUP_TAGS, brackets: true }); + if (!text.includes('<')) return [text, []]; + return splitWithExpr(text, { xmlTags: ALL_MARKUP_TAGS, brackets: false }); } /** * Build the `lk.expression` transcription attribute from stripped markup tags. * * Surfaces a segment's leading delivery/emotion (`expression` for Inworld/xAI, - * `emotion` for Cartesia) as `{"value": ...}` so the frontend can react to it. + * `emotion` for Cartesia) as the provider expression and its normalized mood. * Returns `undefined` when no such tag was present. */ export function expressionAttribute(tags: ExpressiveTag[]): Record | undefined { @@ -881,7 +974,10 @@ export function expressionAttribute(tags: ExpressiveTag[]): Record this.buf.lastIndexOf('>')) { const nxt = this.buf.slice(lastLt + 1, lastLt + 2); @@ -909,7 +1005,7 @@ export class TranscriptMarkupStripper { return true; } } - return this.buf.lastIndexOf('[') > this.buf.lastIndexOf(']'); + return false; } /** Feed a chunk; return the clean text ready to emit (may be empty). */ diff --git a/agents/src/tts/expr_markup.test.ts b/agents/src/tts/expr_markup.test.ts index a60fba1689..a7f5f4f86e 100644 --- a/agents/src/tts/expr_markup.test.ts +++ b/agents/src/tts/expr_markup.test.ts @@ -13,6 +13,7 @@ */ import { describe, expect, it } from 'vitest'; import { DEFAULT_EXPRESSIVE_OPTIONS, resolveExpressiveOptions } from '../voice/agent_session.js'; +import { matchMood } from './_mood.js'; import { TranscriptMarkupStripper, convertMarkup, @@ -132,7 +133,7 @@ describe('Fish Audio dialect', () => { it('registers LLM instructions', () => { const instructions = llmInstructions('fishaudio'); expect(instructions).toBeDefined(); - for (const emotion of [ + const emotions = [ 'regretful', 'hopeful', 'happy', @@ -142,12 +143,27 @@ describe('Fish Audio dialect', () => { 'sad', 'empathetic', 'sarcastic', - ]) { + 'calm', + 'angry', + 'worried', + 'nervous', + 'confident', + 'grateful', + 'delighted', + 'disappointed', + 'frustrated', + 'determined', + ]; + for (const emotion of emotions) { expect(instructions).toContain(emotion); + expect(matchMood(emotion, null), emotion).not.toBeNull(); } expect(instructions).toContain(''); expect(instructions).toContain('clear throat'); expect(instructions).toContain(''); + for (const tone of ['whispering', 'soft', 'shouting', 'hurried']) { + expect(instructions).toContain(tone); + } }); it('converts expr markers to Fish brackets', () => { @@ -169,6 +185,23 @@ describe('Fish Audio dialect', () => { it('converts sound aliases', () => { expect(convertMarkup('fishaudio', '')).toBe('[laughing]'); + expect(convertMarkup('fishaudio', '')).toBe('[sighing]'); + expect(convertMarkup('fishaudio', '')).toBe('[sobbing]'); + }); + + it('converts wrapping and self-closing tones to prefix markers', () => { + expect( + convertMarkup( + 'fishaudio', + `don't tell anyone okay?`, + ), + ).toBe(`[whispering] don't tell anyone okay?`); + expect(convertMarkup('fishaudio', ' hey')).toBe( + '[soft] hey', + ); + expect( + convertMarkup('fishaudio', 'ahoy'), + ).toBe('ahoy'); }); it('splitAllMarkup strips expr markers', () => { @@ -194,6 +227,14 @@ describe('Fish Audio dialect', () => { expect(convertMarkup('fishaudio', raw)).toBe('[very happy] Hey there [emphasis] friend'); }); + it('strips a tone wrapper from the transcript while preserving its words', () => { + const [clean, tags] = splitAllMarkup( + 'We won the whole thing!', + ); + expect(clean).toBe('We won the whole thing!'); + expect(tags).toContainEqual({ type: 'prosody', value: 'shouting' }); + }); + it('filters sounds and examples with steering', () => { let instructions = llmInstructions('fishaudio', { nonverbalSounds: false }); expect(instructions).toBeDefined(); @@ -212,7 +253,11 @@ describe('Fish Audio dialect', () => { it('reports supported nonverbals', () => { expect(supportedNonverbals('fishaudio')).toEqual({ laughing: ['laughing', 'chuckling'], - reflexSounds: ['clear throat'], + breathing: ['gasping'], + sighing: ['sighing'], + crying: ['sobbing'], + vocalizing: ['groaning'], + reflexSounds: ['clear throat', 'yawning'], }); }); @@ -243,7 +288,9 @@ describe('Fish Audio dialect', () => { expect(off).not.toContain('type="sound"'); expect(off).not.toContain('laughing'); for (const instructions of [on, defaultInstructions]) { - expect(instructions).toContain('laughing, chuckling, clear throat'); + expect(instructions).toContain( + 'laughing, chuckling, clear throat, sighing, gasping, groaning, yawning, sobbing', + ); } }); @@ -267,7 +314,9 @@ describe('Fish Audio dialect', () => { const steering = { nonverbalSounds: { laughing: false } }; const fish = llmInstructions('fishaudio', steering); expect(fish).not.toContain('laughing'); - expect(fish).toContain('clear throat'); + for (const kept of ['clear throat', 'sighing', 'gasping', 'groaning', 'yawning', 'sobbing']) { + expect(fish).toContain(kept); + } const inworld = llmInstructions('inworld', steering); expect(inworld).not.toContain('label="laugh"'); @@ -354,10 +403,10 @@ describe('transcript stripping (per-provider + provider-agnostic)', () => { const text = ' Hello! [sigh]'; const [clean, tags] = splitAllMarkup(text); - expect(clean.trim()).toBe('Hello!'); + expect(clean.trim()).toBe('Hello! [sigh]'); expect(tags).toContainEqual({ type: 'expression', value: 'say playfully' }); expect(tags).toContainEqual({ type: 'sound', value: 'laugh' }); - expect(tags).toContainEqual({ type: '', value: 'sigh' }); + expect(tags).not.toContainEqual({ type: '', value: 'sigh' }); }); it('preserves document order when mixing native and expr markup', () => { @@ -370,7 +419,9 @@ describe('transcript stripping (per-provider + provider-agnostic)', () => { { type: 'emotion', value: 'sad' }, { type: 'expression', value: 'happy' }, ]); - expect(expressionAttribute(tags)).toEqual({ 'lk.expression': '{"value":"sad"}' }); + expect(expressionAttribute(tags)).toEqual({ + 'lk.expression': '{"expression":"sad","mood":"sad"}', + }); // same through the per-provider path, with brackets in the mix const [, inworldTags] = splitMarkup( diff --git a/agents/src/tts/markup_utils.test.ts b/agents/src/tts/markup_utils.test.ts index 188f336e24..e0ca352c83 100644 --- a/agents/src/tts/markup_utils.test.ts +++ b/agents/src/tts/markup_utils.test.ts @@ -60,7 +60,7 @@ describe('xAI dialect', () => { expect(instr).toBeDefined(); // this branch instructs the unified expr dialect; convertMarkup lowers it to // xAI's native syntax (see expr_markup.test.ts) - expect(instr).toContain(''); + expect(instr).toContain(''); expect(instr).toContain('Hi there ' + '[pause] friend', ); - expect(clean).toBe('Hi there friend'); + expect(clean).toBe('Hi there [pause] friend'); const types = tags.map((t) => [t.type, t.value]); expect(types).toContainEqual(['emotion', 'happy']); expect(types).toContainEqual(['expression', 'warm']); expect(types).toContainEqual(['sound', 'giggle']); - expect(types).toContainEqual(['', 'pause']); + expect(types).not.toContainEqual(['', 'pause']); }); it('expressionAttribute builds the lk.expression attribute', () => { let [, tags] = splitAllMarkup('oh no'); - expect(expressionAttribute(tags)).toEqual({ 'lk.expression': '{"value":"sad"}' }); + expect(expressionAttribute(tags)).toEqual({ + 'lk.expression': '{"expression":"sad","mood":"sad"}', + }); // no expression/emotion tag -> no attribute (bracket sounds don't count) [, tags] = splitAllMarkup('[pause]hi'); @@ -172,7 +173,15 @@ describe('universal transcript stripping', () => { out += s.flush(); expect(out).not.toContain(' { + const s = new TranscriptMarkupStripper(); + expect(s.push('See [the docs')).toBe('See [the docs'); + expect(s.push('](https://livekit.io).')).toBe('](https://livekit.io).'); }); it('TranscriptMarkupStripper does not stall on a bare "<"', () => { diff --git a/agents/src/tts/mood.test.ts b/agents/src/tts/mood.test.ts new file mode 100644 index 0000000000..8ca187aaf3 --- /dev/null +++ b/agents/src/tts/mood.test.ts @@ -0,0 +1,39 @@ +// SPDX-FileCopyrightText: 2026 LiveKit, Inc. +// +// SPDX-License-Identifier: Apache-2.0 +import { describe, expect, it } from 'vitest'; +import { DEFAULT_MOOD, MOOD_KEYWORDS, MOOD_PRIORITY, matchMood } from './_mood.js'; +import { expressionAttribute, splitAllMarkup } from './_provider_format.js'; + +describe('mood normalization', () => { + it('prioritizes every mood exactly once', () => { + expect(new Set(MOOD_PRIORITY)).toEqual(new Set(Object.keys(MOOD_KEYWORDS))); + expect(MOOD_PRIORITY).toHaveLength(new Set(MOOD_PRIORITY).size); + }); + it.each([ + ['excited', 'excited'], + ['soft, with genuine care', 'empathetic'], + ['bright, upbeat energy', 'excited'], + ['gently curious, welcoming', 'curious'], + ['furious', 'angry'], + ['Excited!!!', 'excited'], + ] as const)('matches %s as %s', (label, expected) => expect(matchMood(label)).toBe(expected)); + it.each(['like a pirate', 'across the room', '', ' '])('falls back for %s', (label) => { + expect(matchMood(label)).toBe(DEFAULT_MOOD); + expect(matchMood(label, null)).toBeNull(); + }); + it('matches keywords only at word starts', () => { + expect(matchMood('irate', null)).toBe('angry'); + expect(matchMood('pirate', null)).toBeNull(); + expect(matchMood('excited', null)).toBe('excited'); + expect(matchMood('unexcited', null)).toBeNull(); + }); + it.each([ + ['soft, with genuine care', 'empathetic'], + ['like a pirate', DEFAULT_MOOD], + ] as const)('carries expression %s and mood %s', (label, mood) => { + const [, tags] = splitAllMarkup(`hey`); + const attr = expressionAttribute(tags); + expect(JSON.parse(attr!['lk.expression']!)).toEqual({ expression: label, mood }); + }); +}); diff --git a/agents/src/voice/agent_session.test.ts b/agents/src/voice/agent_session.test.ts index bf4ec27ab5..ccc81a34c6 100644 --- a/agents/src/voice/agent_session.test.ts +++ b/agents/src/voice/agent_session.test.ts @@ -6,6 +6,13 @@ import { AgentSession, resolveRecordingOptions } from './agent_session.js'; import { SpeechHandle } from './speech_handle.js'; describe('AgentSession.run', () => { + it('defaults expressive off and accepts boolean or options', () => { + expect(new AgentSession().sessionOptions.expressive).toBe(false); + expect(new AgentSession({ expressive: true }).sessionOptions.expressive).toBe(true); + const expressive = { ttsInstructionsAppend: 'Stay upbeat.' }; + expect(new AgentSession({ expressive }).sessionOptions.expressive).toEqual(expressive); + }); + it('forwards inputModality to generateReply', async () => { const session = new AgentSession(); const generateReply = vi diff --git a/agents/src/voice/index.ts b/agents/src/voice/index.ts index cf91a84c58..4195c12cc6 100644 --- a/agents/src/voice/index.ts +++ b/agents/src/voice/index.ts @@ -19,6 +19,7 @@ export { AgentSession, type AgentSessionOptions, type AgentSessionUsage, + type ExpressiveOptions, type NonverbalOptions, type SpeechSteeringOptions, type VoiceOptions, diff --git a/agents/src/voice/room_io/_output.ts b/agents/src/voice/room_io/_output.ts index 316d2c708c..1dea7c5395 100644 --- a/agents/src/voice/room_io/_output.ts +++ b/agents/src/voice/room_io/_output.ts @@ -211,7 +211,7 @@ export class ParticipantTranscriptionOutput extends BaseParticipantTranscription if (this.isDeltaStream) { // reuse the existing writer if (this.writer === null) { - this.writer = await this.createTextWriter(); + this.writer = await this.createTextWriter(expressionAttribute(this.stripper.tags)); } await this.writer.write(payload); } else { @@ -250,9 +250,8 @@ export class ParticipantTranscriptionOutput extends BaseParticipantTranscription const currWriter = this.writer; this.writer = null; - // only emit on a segment that captured text (keeps lk.transcription cadence intact). - // The leading expression the sinks stripped rides along on the closing chunk as the - // lk.expression attribute. + // Only emit on a segment that captured text. Delta streams attach the leading expression + // when opening the writer because rtc-node close() cannot carry attributes. let remaining: string; let tags: ExpressiveTag[]; if (this.isDeltaStream) { @@ -278,13 +277,12 @@ export class ParticipantTranscriptionOutput extends BaseParticipantTranscription throw new Error('localParticipant not found'); } - if (!attributes) { - attributes = { - [ATTRIBUTE_TRANSCRIPTION_FINAL]: 'false', - }; - if (this.trackId) { - attributes[ATTRIBUTE_TRANSCRIPTION_TRACK_ID] = this.trackId; - } + attributes = { + [ATTRIBUTE_TRANSCRIPTION_FINAL]: 'false', + ...attributes, + }; + if (this.trackId) { + attributes[ATTRIBUTE_TRANSCRIPTION_TRACK_ID] = this.trackId; } attributes[ATTRIBUTE_TRANSCRIPTION_SEGMENT_ID] = this.currentId; diff --git a/examples/src/expressive_agent.ts b/examples/src/expressive_agent.ts index 3d5b1d3b71..c54eb89a30 100644 --- a/examples/src/expressive_agent.ts +++ b/examples/src/expressive_agent.ts @@ -1,8 +1,6 @@ // SPDX-FileCopyrightText: 2026 LiveKit, Inc. // // SPDX-License-Identifier: Apache-2.0 -// cue-cli e2e harness agent for expressive mode (expr marker dialect). -// Registered with explicit dispatch as `expressive-agent-js`. import { Agent, AgentSession, @@ -11,40 +9,53 @@ import { cli, defineAgent, inference, - tool, + log, } from '@livekit/agents'; +import { readFileSync } from 'node:fs'; import { fileURLToPath } from 'node:url'; -import { z } from 'zod'; +import { parseSessionConfig } from './expressive_agent/protocol.ts'; + +const instructions = readFileSync( + new URL('../src/expressive_agent/prompt.md', import.meta.url), + 'utf8', +).replace(/^\s*/, ''); + +const greeting = + "Open the call the way you'd answer the phone to someone you know well. " + + "Short and warm, and leave them room to say what's going on."; + +class Friend extends Agent { + constructor() { + super({ instructions }); + } + + async onEnter(): Promise { + this.session.generateReply({ instructions: greeting }); + } +} export default defineAgent({ entry: async (ctx: JobContext) => { - const agent = Agent.create({ - instructions: - 'You are a cheerful, expressive assistant. Keep replies to one or two short ' + - 'sentences. You can hear the user and respond with speech.', - tools: [ - tool({ - name: 'getWeather', - description: 'Get the weather for a given location.', - parameters: z.object({ - location: z.string().describe('The location to get the weather for'), - }), - execute: async ({ location }) => `The weather in ${location} is sunny.`, - }), - ], - }); + const config = parseSessionConfig(ctx.job.metadata); + log().info( + { expressive: config.expressive, voice: config.voice.label }, + 'starting expressive session', + ); const session = new AgentSession({ - stt: new inference.STT({ model: 'deepgram/nova-3', language: 'en' }), - llm: new inference.LLM({ model: 'openai/gpt-4.1-mini' }), - // Inworld: free-form expression labels (with spaces) — exercises both the - // expr dialect lowering and the transcript pacing fix. - tts: new inference.TTS({ model: 'inworld/inworld-tts-2' }), - expressive: true, + stt: new inference.STT({ model: 'assemblyai/universal-3-5-pro', language: 'en' }), + llm: new inference.LLM({ model: 'google/gemma-4-31b-it' }), + tts: new inference.TTS({ model: config.voice.model, voice: config.voice.voice }), + turnHandling: { + turnDetection: new inference.TurnDetector({ version: 'v1' }), + interruption: { mode: 'adaptive' }, + preemptiveGeneration: { enabled: true }, + }, + expressive: config.expressive, }); - await session.start({ agent, room: ctx.room }); - session.say('Hi there! How can I help you today?'); + await session.start({ agent: new Friend(), room: ctx.room }); + await ctx.room.localParticipant?.setAttributes(config.attributes()); }, }); diff --git a/examples/src/expressive_agent/README.md b/examples/src/expressive_agent/README.md new file mode 100644 index 0000000000..3450b7f1f4 --- /dev/null +++ b/examples/src/expressive_agent/README.md @@ -0,0 +1,31 @@ + + +# Expressive Agent + +A free-form voice agent that demonstrates Expressive Mode. There is no task or +tool: talk to it like a friend and it adapts its delivery to the conversation. + +The pipeline uses LiveKit Inference with AssemblyAI Universal 3.5 Pro, Gemma 4 +31B, Fish Audio S2.1 Pro by default, and the LiveKit cloud turn detector. It +also enables adaptive interruption handling and preemptive generation. + +## Configuration + +Dispatch metadata can select the voice and toggle expressive output: + +```json +{ "expressive": true, "tts": "fishaudio" } +``` + +Supported `tts` values are `fishaudio`, `inworld`, `cartesia`, and `xai`. +Malformed metadata falls back to all defaults; an unknown voice falls back to +Fish Audio while preserving the requested expressive setting. The active +settings are published as the agent participant attributes `expressive`, +`tts_provider`, and `tts_label`. + +The explicit worker name remains `expressive-agent-js`, including for cue-cli +sessions. diff --git a/examples/src/expressive_agent/prompt.md b/examples/src/expressive_agent/prompt.md new file mode 100644 index 0000000000..8b2fd2ec72 --- /dev/null +++ b/examples/src/expressive_agent/prompt.md @@ -0,0 +1,43 @@ + + +You are the user's closest friend, catching up over a call. There is no task +here, no ticket to close, no form to fill. You are just talking. + +Expressive Mode injects the delivery guide separately, so this prompt covers +only who you are and what you say. Tone and pacing rules don't belong here, but +word choice does. + +# Output rules + +- One or two sentences. Three is already too many. +- Plain prose. No markdown, lists, bullets, headers, or emojis. +- Spell out numbers, money, and dates as you would say them out loud. +- Use contractions. "It's", not "it is"; "you're", not "you are". +- Never use input vocabulary like "enter" or "fill in". They are speaking, not typing. + +# How you talk + +- Speak naturally, not from a customer-service script. You are not assisting + anyone, you are talking with them. +- Don't open two consecutive turns with the same word. +- React before you respond. If they tell you something big, the reaction comes + first and the follow-up question comes second. +- Ask about one thing at a time, the way a friend would, not the way a survey does. +- Trust their memory. They heard what you said five seconds ago, so don't restate it. +- If they interrupted you, don't restart the sentence. What they said is the new subject. +- When they are venting, stay on their side. Advice they didn't ask for is worth + less than agreeing that something sucks. +- Don't reach for a silver lining they didn't ask for, and don't rush to fix + what they only wanted to say out loud. +- Never explain or narrate your own tone. + +# Guardrails + +- You have no name unless they give you one, and you never introduce yourself by one. +- You are a friend, not a therapist or a doctor. If they raise something that + needs real help, say plainly that you are worried and that this is worth + talking to someone about. Don't lecture, and don't pretend to be qualified. diff --git a/examples/src/expressive_agent/protocol.ts b/examples/src/expressive_agent/protocol.ts new file mode 100644 index 0000000000..8553b823b0 --- /dev/null +++ b/examples/src/expressive_agent/protocol.ts @@ -0,0 +1,85 @@ +// SPDX-FileCopyrightText: 2026 LiveKit, Inc. +// +// SPDX-License-Identifier: Apache-2.0 +import { log } from '@livekit/agents'; +import { z } from 'zod'; + +const DEFAULT_VOICE = 'fishaudio'; + +interface Voice { + provider: string; + model: string; + voice: string; + label: string; +} + +const VOICES = { + fishaudio: { + provider: 'fishaudio', + model: 'fishaudio/s2.1-pro', + voice: '51b44863613e405a896f7f4294c6e6d0', + label: 'Fish Audio S2.1 Pro (Marley)', + }, + inworld: { + provider: 'inworld', + model: 'inworld/inworld-tts-2', + voice: 'Ashley', + label: 'Inworld TTS 2 (Ashley)', + }, + cartesia: { + provider: 'cartesia', + model: 'cartesia/sonic-3', + voice: '9626c31c-bec5-4cca-baa8-f8ba9e84c8bc', + label: 'Cartesia Sonic 3 (Jacqueline)', + }, + xai: { + provider: 'xai', + model: 'xai/tts-1', + voice: 'eve', + label: 'xAI TTS 1 (Eve)', + }, +} as const satisfies Record; + +const boolSchema = z.preprocess((value) => { + if (value === 'true' || value === 1) return true; + if (value === 'false' || value === 0) return false; + return value; +}, z.boolean()); + +const requestSchema = z.object({ + expressive: boolSchema.default(true), + tts: z.string().optional(), +}); + +export interface SessionConfig { + expressive: boolean; + voice: Voice; + attributes(): Record; +} + +export function parseSessionConfig(metadata?: string): SessionConfig { + let request: z.infer = { expressive: true }; + + if (metadata) { + try { + const result = requestSchema.safeParse(JSON.parse(metadata)); + if (!result.success) throw result.error; + request = result.data; + } catch { + log().warn({ metadata }, 'ignoring malformed expressive-agent dispatch metadata'); + } + } + + const requestedVoice = request.tts as keyof typeof VOICES | undefined; + const voice = VOICES[requestedVoice ?? DEFAULT_VOICE] ?? VOICES[DEFAULT_VOICE]; + + return { + expressive: request.expressive, + voice, + attributes: () => ({ + expressive: request.expressive ? 'true' : 'false', + tts_provider: voice.provider, + tts_label: voice.label, + }), + }; +}