1(function () { 2 'use strict'; 3 4 /** 5 * @param {object} config 6 * @param {string} [config.service='chat'] 7 * @param {string|null|undefined} [config.lang] â omit for a UI-language mic (resolved 8 * fresh per session via resolveSttUiLang), an explicit code ('da') for a fixed 9 * content-language mic (translator source, editor dictation), null for auto-detect 10 * @param {string} [config.tokenEndpoint] â token service override; falls back to the 11 * page-level window.__SS_STT_TOKEN_ENDPOINT and then /api/stt-token.php 12 * @param {Function} config.onTranscript â (text: string) for each final transcript 13 * @param {Function} [config.onPartial] â (text: string) for interim results 14 * @param {Function} [config.onAudioActivity] â (active: boolean, info: object) 15 * local microphone voice activity, independent of transcript/network cadence 16 * @param {Function} [config.onError] â (message: string) 17 * @param {Function} [config.onStateChange] â (state: 'idle'|'connecting'|'listening'|'error') 18 * @returns {{ start, stop, toggle, destroy, isActive, setLang }} 19 */ 20 let _toastEl$1 = null; 21 let _toastTimer$1 = null; 22 23 // Lightweight local VAD for endpointing. Transcript updates can pause while a 24 // person is still speaking (provider buffering/network delay), so consumers 25 // need a second signal sourced directly from captured microphone PCM. The 26 // adaptive floor follows quiet background noise; the absolute floor prevents 27 // near-silence from becoming "voice", and the hangover bridges gaps between 28 // syllables. It never controls capture or provider traffic â it only reports. 29 function createVoiceActivityDetector(onActivity, options = {}) { 30 const minRms = Number(options.minRms) > 0 ? Number(options.minRms) : 0.012; 31 const noiseMultiplier = Number(options.noiseMultiplier) > 1 ? Number(options.noiseMultiplier) : 3; 32 const hangoverMs = Number(options.hangoverMs) >= 0 ? Number(options.hangoverMs) : 300; 33 const emitIntervalMs = Number(options.emitIntervalMs) > 0 ? Number(options.emitIntervalMs) : 100; 34 let noiseFloor = 0.003; 35 let voiceUntil = 0; 36 let lastEmitAt = 0; 37 let wasActive = false; 38 39 function process(samples, now = Date.now()) { 40 if (!samples || !samples.length) return false; 41 let sumSquares = 0; 42 for (let i = 0; i < samples.length; i++) { 43 const normalized = samples[i] / 32768; 44 sumSquares += normalized * normalized; 45 } 46 const rms = Math.sqrt(sumSquares / samples.length); 47 const threshold = Math.max(minRms, Math.min(0.08, noiseFloor * noiseMultiplier)); 48 const instantVoice = rms >= threshold; 49 50 if (instantVoice) { 51 voiceUntil = now + hangoverMs; 52 } else { 53 // Learn the quiet floor slowly and cap it so sustained environmental 54 // noise cannot raise the speech threshold without bound. 55 noiseFloor = Math.min(0.025, noiseFloor * 0.98 + rms * 0.02); 56 } 57 58 const active = instantVoice || now < voiceUntil; 59 if (typeof onActivity === 'function') { 60 if (active && (!wasActive || now - lastEmitAt >= emitIntervalMs)) { 61 lastEmitAt = now; 62 onActivity(true, { rms, threshold }); 63 } else if (!active && wasActive) { 64 onActivity(false, { rms, threshold }); 65 } 66 } 67 wasActive = active; 68 return active; 69 } 70 71 function reset() { 72 noiseFloor = 0.003; 73 voiceUntil = 0; 74 lastEmitAt = 0; 75 wasActive = false; 76 } 77 78 return { process, reset }; 79 } 80 81 function _sttShortLang(value) { 82 if (!value) return null; 83 const short = String(value).trim().toLowerCase().replace(/_/g, '-').split('-')[0]; 84 return /^[a-z]{2,3}$/.test(short) ? short : null; 85 } 86 87 // The platform-wide answer to "which language should a UI mic listen in" (#3781 88 // follow-up: email read the legacy ss_chat_lang key, ugeplan and the help form were 89 // hardcoded 'da' â four robots, four resolvers). Precedence: the live chat i18n 90 // state, then the account-menu/product choice (sb_ui_lang â lang-selector.js also 91 // stamps documentElement.lang from it at boot), then the legacy speech-language key 92 // (still written by mobile-nav and the auth pages), then the served <html lang>. 93 // The server (api/stt-token.php) canonicalises against its own allowlist, so an 94 // unsupported code degrades to provider auto-detect there, never to a 4xx.
95 function resolveSttUiLang() { 96 try { 97 if (typeof window !== 'undefined' && window.ChatI18n && typeof window.ChatI18n.getLang === 'function') { 98 const l = _sttShortLang(window.ChatI18n.getLang()); 99 if (l) return l; 100 } 101 } catch (e) {} 102 try { 103 const l = _sttShortLang(localStorage.getItem('sb_ui_lang')); 104 if (l) return l; 105 } catch (e) {} 106 try { 107 const s = localStorage.getItem('ss_chat_lang'); 108 if (s) { 109 const o = JSON.parse(s); 110 const l = _sttShortLang(o && o.code); 111 if (l) return l; 112 } 113 } catch (e) {} 114 try { 115 if (typeof document !== 'undefined') { 116 const l = _sttShortLang(document.documentElement.lang); 117 if (l) return l; 118 } 119 } catch (e) {} 120 return 'da'; 121 } 122 function _showSttToast$1(msg, duration = 4000) { 123 if (!_toastEl$1) { 124 _toastEl$1 = document.createElement('div'); 125 _toastEl$1.className = 'ss-stt-toast'; 126 _toastEl$1.setAttribute('role', 'alert'); 127 _toastEl$1.setAttribute('aria-live', 'assertive'); 128 document.body.appendChild(_toastEl$1); 129 } 130 _toastEl$1.textContent = msg; 131 _toastEl$1.classList.add('ss-stt-toast--visible'); 132 clearTimeout(_toastTimer$1); 133 _toastTimer$1 = setTimeout(() => { 134 _toastEl$1.classList.remove('ss-stt-toast--visible'); 135 }, duration); 136 } 137 138 // ââ Danish dictation post-format (#5734) ââââââââââââââââââââââââââââââââââââ 139 // The STT provider's smart-format layer is English-centric: on Danish it emits 140 // anglo decimal points ("2.5"), the Spanish ordinal glyph ("1º"), the misspelling 141 // "hundred" (da: "hundrede"), a capitalized preposition "I" (English-I habit) 142 // and capitalized month/time words after abbreviation periods. This is the 143 // deterministic last-mile mirror of the TTS-side normalizeSpeechText() in 144 // assets/js/shared/tts-engine.js â SAME Danish conventions, opposite direction 145 // (dictation output â document text), so text dictated here is later read 146 // aloud correctly by that rail (its halvanden-rule requires comma decimals). 147 // Orthography ONLY: rules never change digit values, names, or negations. 148 // Applied solely when the session's effective STT language is 'da'. 149 150 // Words that (lowercased) directly follow the PREPOSITION "i" â never the 151 // plural-you pronoun "I". Months included ("i august"); bare digits are 152 // handled in the rule itself ("i 2026"). 153 // STRONG: temporal words and fixed "i â¦"-phrases that NEVER follow the 154 // pronoun â they beat every other signal ("Der er i alt fem"). 155 const STT_DA_I_LOWER_NEXT = new Set([ 156 'dag', 'morgen', 'gÃ¥r', 'gaar', 'aften', 'nat', 'morges', 'aftes', 'fjor', 157 'weekend', 'weekenden', 'Ã¥r', 'Ã¥ret', 'uge', 'ugen', 'uger', 'mÃ¥ned', 158 'mÃ¥neden', 'mÃ¥neder', 'sommer', 'sommers', 'vinter', 'vinters', 'efterÃ¥r', 159 'efterÃ¥ret', 'forÃ¥r', 'forÃ¥ret', 'starten', 'slutningen', 'begyndelsen', 160 'midten', 'løbet', 'stedet', 'gang', 'gangen', 'forbindelse', 'forhold', 161 'øjeblikket', 'mellemtiden', 'nærheden', 'orden', 'tvivl', 'stand', 162 'alt', 'aftenen', 163 'januar', 'februar', 'marts', 'april', 'maj', 'juni', 'juli', 'august', 164 'september', 'oktober', 'november', 'december' 165 ]); 166 167 // WEAK: determiners that are common after BOTH readings ("i denne uge" / 168 // "Kan I det?") â they lower only when no pronoun signal protects (hostile 169 // r6: checking these before the prev-word signal corrupted "Kan I det?"). 170 const STT_DA_I_LOWER_WEAK_NEXT = new Set([ 171 'denne', 'dette', 'den', 'det', 'de', 'en', 'et', 'ét', 'hele', 172 'hvert', 'hver', 'sidste', 'første', 'næste' 173 ]); 174 175 // Words that follow the PRONOUN "I" (finite verbs, common infinitives after a 176 // modal, and post-pronoun function words) â the capital must survive. Checked 177 // after the lower-list, so "i gÃ¥r"/"i dag" win over the rare "I gÃ¥r ned". 178 const STT_DA_I_KEEP_NEXT = new Set([ 179 'skal', 'kan', 'vil', 'mÃ¥', 'maa', 'bør', 'kunne', 'skulle', 'ville', 180 'burde', 'har', 'havde', 'er', 'var', 'bliver', 'blev', 'fÃ¥r', 'fik', 181 'gør', 'gjorde', 'kommer', 'kom', 'ser', 'sÃ¥', 'saa', 'tager', 'tog', 182 'siger', 'sagde', 'ved', 'vidste', 'hedder', 'hed', 'ønsker', 'ønskede', 183 'behøver', 'sender', 'sendte', 'ringer', 'ringede', 'skriver', 'skrev', 184 'læser', 'læste', 'spørger', 'spurgte', 'svarer', 'svarede', 'betaler', 185 'betalte', 'mangler', 'finder', 'fandt', 'bruger', 'brugte', 'kender', 186 'kendte', 'mødes', 'mødtes', 'ses', 'arbejder', 'bor', 'boede', 'husker', 187 'glemmer', 'vælger', 'valgte', 'holder', 'holdt', 'giver', 'gav', 'tror', 188 'hÃ¥ber', 'hjælpe', 'sende', 'ringe', 'komme', 'se', 'give', 'finde', 189 'fortælle', 'kontakte', 'oplyse', 'bekræfte', 'svare', 'vente', 'betale',
190 'deltage', 'huske', 'glemme', 'Ã¥bne', 'lukke', 'læse', 'skrive', 'gøre', 191 'tage', 'fÃ¥', 'faa', 'være', 'blive', 'holde', 'bruge', 'vælge', 192 'undersøge', 'vurdere', 'behandle', 'godkende', 'afvise', 'sørge', 193 // Finite present/past forms common in letters TO an authority ("hÃ¥ber I 194 // modtagerâ¦", "det I lovede", "som I nævnte") â lens-found 12/7. 195 'modtager', 'behandler', 'godkender', 'afviser', 'kontakter', 'bekræfter', 196 'oplyser', 'undersøger', 'vurderer', 'forstÃ¥r', 'hører', 'mener', 197 'accepterer', 'ignorerer', 'lovede', 'nævnte', 'besluttede', 'nægtede', 198 'tildelte', 'anmodede', 'modtog', 'afgjorde', 'indsendte', 'overførte', 199 'indgik', 200 'til', 'om', 'med', 'fra', 'uden', 'ikke', 'alle', 'begge', 'selv', 201 'ogsÃ¥', 'nu', 'jo', 'bare', 'her', 'der', 'dog', 'vel', 'altsÃ¥', 202 'gerne', 'godt', 'snart', 'stadig', 'allerede', 'sammen', 203 // Hostile r1: pronoun continuations â "Det er I somâ¦", "Kan I to komme?". 204 'som', 'to', 'tre', 'fire', 'fem', 'seks' 205 ]); 206 207 // Words BEFORE a capital "I" that signal the PRONOUN (hostile r2: 208 // verb-first inversion "Er I klar?", "Bliver I færdige?" and subordinating 209 // conjunctions "som/at/hvis Iâ¦"). Adjective continuations are an open class, 210 // so the NEXT-word lists can never close them â the PREV word is the 211 // structural signal. Precedence in the rule: strong preposition objects 212 // (next) beat this list ("Der er i alt fem"). 213 const STT_DA_I_PRON_PREV = new Set([ 214 'er', 'var', 'bliver', 'blev', 'kan', 'kunne', 'skal', 'skulle', 'vil', 215 'ville', 'mÃ¥', 'maa', 'bør', 'burde', 'har', 'havde', 'fÃ¥r', 'fik', 216 'gør', 'gjorde', 'kommer', 'kom', 'gÃ¥r', 'gik', 'siger', 'sagde', 217 'tror', 'troede', 'hÃ¥ber', 'hÃ¥bede', 'ved', 'vidste', 'mener', 'mente', 218 'synes', 'syntes', 'ønsker', 'ønskede', 'gider', 'tør', 'behøver', 219 'plejer', 'sender', 'henter', 'venter', 'møder', 'besøger', 'hjælper', 220 'køber', 'leverer', 'afleverer', 'tilbyder', 'anbefaler', 221 'hvornÃ¥r', 'hvordan', 'hvorfor', 'hvad', 'hvem', 222 'som', 'at', 'hvis', 'nÃ¥r', 'naar', 'da', 'mens', 'fordi', 'inden', 223 'før', 'foer', 'efter', 'om' 224 ]); 225 226 // Words BEFORE a capital "I" that mark it a ROMAN NUMERAL / label ("Type I 227 // diabetes", "Kapitel I", "Del I") â never the preposition (hostile r8). 228 const STT_DA_I_ROMAN_PREV = new Set([ 229 'kapitel', 'del', 'type', 'gruppe', 'klasse', 'model', 'bind', 'akt', 230 'fase', 'grad', 'niveau', 'afsnit', 'sektion', 'trin' 231 ]); 232 233 // R1 context guard (hostile r1): a dot between digits right after an 234 // identifier word is a version/section id, not a Danish decimal. Unicode 235 // letter boundary in front (hostile r2: a bare `v$` alternative would match 236 // "brev"/"blev", and `punkt` would match "tidspunkt"). 237 const STT_DA_DECIMAL_ID_PREFIX = 238 /(?:^|[^\p{L}])(?:version|punkt|afsnit|kapitel|stk\.?|pkt\.?|ios|android|gpt-?\d*|v)\s*$/iu; 239 240 // Quantity-context words BEFORE a dotted number mark it a decimal even in 241 // date shape ("vejer 3.5", "pÃ¥ 2.5") â dates never follow these (hostile r5). 242 const STT_DA_DECIMAL_QTY_PREFIX = 243 /(?:^|[^\p{L}])(?:vejer|vejede|koster|kostede|mÃ¥ler|mÃ¥lte|sparer|betaler|pÃ¥|omkring|cirka|ca\.?)\s*$/iu; 244 245 // DATE context in the trailing words before a dotted number ("Fristen er 246 // 31.5", "mødet er 3.5", "den 28.5"). Hostile r9 flipped the precedence: 247 // bare decimals now CONVERT by default and only explicit date context 248 // preserves the dot â "2.5" alone and "Jeg fik 7.5" (school grades!) are 249 // decimals, date shapes without date words were over-protected. 250 const STT_DA_DATE_CONTEXT_BEFORE = 251 /(?:^|[^\p{L}])(?:frist|møde|dato|deadline|aflever|senest|inden|den|d\.|periode|kvartal|uge)[\p{L}]*[^\p{L}\n]+(?:[\p{L}'â\-]+[^\p{L}\n]+){0,2}$/iu; 252 253 // ONE shared unit regex for both decimal rules (hostile r13: ASCII \b after 254 // the one-letter units g/m/l matched the m|ø seam in "mødes" â Danish 255 // letters are non-word to \b â so "Kl. 14.30 mødes" converted the clock). 256 // Unicode-aware boundary instead; defined once so R1/R1b can never drift. 257 const STT_DA_UNIT_FOLLOWS = 258 /^\s?(?:(?:kilo|kg|gram|g|kroner|kr\.?|procent|meter|m|liter|l|km|cm|mm|ml|dl|point|kalorier|Ã¥r|aar|mÃ¥ned|mÃ¥neder|maaned|maaneder|uge|uger|time|timer|minut|minutter|sekund|sekunder)(?=$|[^\p{L}\p{N}_])|%)/iu; 259 260 function normalizeSttDanishText(input) { 261 let text = String(input || ''); 262 if (!text) return text; 263 264 // R2 â ordinal glyph: "1º"/"1ª" (U+00BA/U+00AA) â "1.". The degree sign
265 // (° U+00B0, temperatures) is a different codepoint and stays untouched. 266 // A LOOKALIKE ordinal glyph in temperature position ("30 ºC") is the 267 // degree sign misrecognized â normalize it to real ° (hostile r8; merely 268 // preserving the wrong glyph blessed a visible typo). Runs BEFORE the 269 // decimal rule so the pass stays idempotent (lens 12/7). 270 text = text.replace(/(\d\s?)[ºª](?=\s*[CFcf]\b)/g, '$1°'); 271 // Swallow an immediately following period so a sentence-final ordinal 272 // ("Det er 1º.") becomes "1." not "1.." (hostile r14, double period). 273 text = text.replace(/(\d)\s?[ºª](?!\s*[CFcf]\b)\.?/g, '$1.'); 274 275 // R1 â decimal comma: "2.5" â "2,5" (single-digit fraction here; R1b below 276 // handles two-digit fractions with hour/date guards). Danish thousands 277 // ("1.000") and dotted sequences ("1.2.3") are never touched â but a 278 // sentence period straight after the decimal ("Vejer 2.5.") must not block 279 // it (lens 12/7): only a digit, or a dot that starts another number, 280 // blocks. Digit values themselves are never altered. 281 // (Leading-context group instead of a lookbehind: a lookbehind literal 282 // throws at parse time in older Safari and would kill the whole module.) 283 // Identifier contexts ("version 2.5", "punkt 2.5", "iOS 17.4") keep their 284 // dot â hostile r1: converting an identifier is a semantic change. 285 text = text.replace(/(^|[^\d.])(\d+)\.(\d)(?!\d)(?!\.\d)/g, 286 (full, lead, int, frac, off, whole) => { 287 if (STT_DA_DECIMAL_ID_PREFIX.test(whole.slice(0, off + lead.length))) return full; 288 const unitFollows = STT_DA_UNIT_FOLLOWS.test(whole.slice(off + full.length)); 289 if (unitFollows) return lead + int + ',' + frac; 290 const beforeText = whole.slice(0, off + lead.length); 291 if (STT_DA_DECIMAL_QTY_PREFIX.test(beforeText)) return lead + int + ',' + frac; 292 // Date guard, context-required (hostile r9 flipped r5's precedence): 293 // a bare decimal CONVERTS by default ("2.5", "Jeg fik 7.5"); only a 294 // date shape WITH date words in the trailing context keeps its dot 295 // ("Fristen er 31.5", "den 28.5", "Perioden 2026.7"). 296 const intNum = parseInt(int, 10); 297 const dateShape = (intNum >= 1 && intNum <= 31 && frac !== '0') 298 || (/^(?:1[89]|2[01])\d{2}$/.test(int) && frac !== '0'); 299 if (dateShape && STT_DA_DATE_CONTEXT_BEFORE.test(beforeText)) return full; 300 return lead + int + ',' + frac; 301 }); 302 303 // R1b â two-decimal quantities ("2.75 kilo", "199.95 kroner") are real 304 // dictated Danish (hostile r1). Clock times survive: convert only when the 305 // integer part cannot be an hour (>= 25) or a unit word follows. 306 text = text.replace(/(^|[^\d.])(\d+)\.(\d{2})(?!\d)(?!\.\d)/g, 307 (full, lead, int, frac, off, whole) => { 308 if (STT_DA_DECIMAL_ID_PREFIX.test(whole.slice(0, off + lead.length))) return full; 309 const unitFollows = STT_DA_UNIT_FOLLOWS.test(whole.slice(off + full.length)); 310 if (unitFollows) return lead + int + ',' + frac; 311 const beforeText = whole.slice(0, off + lead.length); 312 if (STT_DA_DECIMAL_QTY_PREFIX.test(beforeText)) return lead + int + ',' + frac; 313 // Context-required guards (hostile r9 â bare "12.99"/"24.50" are 314 // decimals): a dotted CLOCK survives only with time words before it 315 // ("vi mødes 14.30", "kampen slutter 23.59"); a day.month/year.month 316 // DATE survives only with date words before it ("Fristen er 31.12"). 317 const intNum = parseInt(int, 10); 318 const fracNum = parseInt(frac, 10); 319 // A valid HH.MM shape (hour 0-23, minute 0-59) is a clock by default â 320 // "Vi ses 14.30" needs no time word before it (hostile r14). A unit or 321 // quantity prefix already forced conversion above, so a real decimal 322 // that happens to look like a clock ("pÃ¥ 14.30") still converts. 323 const clockShape = intNum <= 23 && fracNum <= 59; 324 if (clockShape) return full; 325 const dateShape = fracNum >= 1 && fracNum <= 12 326 && (intNum <= 31 || /^(?:1[89]|2[01])\d{2}$/.test(int)); 327 if (dateShape && STT_DA_DATE_CONTEXT_BEFORE.test(beforeText)) return full; 328 return lead + int + ',' + frac; 329 }); 330 331 // R3 â spelling: standalone "hundred" is not a Danish word ("hundrede"). 332 // \b is NOT unicode-aware in JS â it would fire INSIDE compounds at æ/ø/Ã¥ 333 // seams ("hundredÃ¥rsdag" â "hundredeÃ¥rsdag", a meaning change; lens 12/7). 334 // Unicode letter context on both sides instead (leading group, no lookbehind). 335 // Digits/underscore count as identifier characters too (hostile r9: 336 // "2hundred"/"_hundred" must stay) â letters alone were too narrow. 337 text = text.replace(/(^|[^\p{L}\p{N}_])([Hh])undred(?![\p{L}\p{N}_])/gu, 338 (full, lead, h) => lead + h + 'undrede'); 339 340 // R4 â the capitalized preposition "I". Mid-sentence only (a sentence may 341 // correctly start with either reading). Two-list decision on the next word: 342 // strong preposition-objects lowercase, pronoun-context words keep. Unknown 343 // next words lowercase â the provider capitalizes every Danish "i", and the 344 // preposition dominates dictated text by an order of magnitude. Known 345 // limitation (documented in the contract): a pronoun before an unlisted 346 // word ("ringer I tilâ¦" inversion) lowercases; cosmetic, never semantic. 347 // Shared capital-I decision (hostile r4 refactor: the sentence-interior 348 // pass and the abbreviation-dot pass must judge identically). 349 // Returns true when the capital must be LOWERED to the preposition. 350 const lowerCapitalI = (next, before, allowFirstWordInversion) => {
351 const key = next.toLocaleLowerCase('da-DK'); 352 // Digits: only a year is preposition-proof ("født I 1995"); a small 353 // count is a pronoun phrase ("Kan I 2 komme?") â hostile r1. 354 if (/^\d/.test(next)) return /^(?:1[89]|2[01])\d{2}$/.test(next); 355 // Precedence (hostile r6): STRONG fixed phrases beat everything ("Der 356 // er i alt fem") â but the WEAK determiner set must lose to the pronoun 357 // signal ("Kan I det?", "Har I en plan?" â determiners follow both 358 // readings), so the prev-word check sits between the two lists. 359 const prevMatch = before.match(/([\p{L}\d'â\-]+)[^\p{L}\d]*$/u); 360 // Roman-numeral labels ("Type I diabetes", "Kapitel I") beat even the 361 // strong list â the I is part of a NAME (hostile r8). 362 if (prevMatch && STT_DA_I_ROMAN_PREV.has(prevMatch[1].toLocaleLowerCase('da-DK'))) return false; 363 if (STT_DA_I_LOWER_NEXT.has(key)) return true; 364 if (prevMatch && STT_DA_I_PRON_PREV.has(prevMatch[1].toLocaleLowerCase('da-DK'))) return false; 365 // Question/inversion generalized (hostile r3): when the word before 366 // "I" is the SENTENCE'S FIRST word, the shape is "Verb I â¦?" ("Ringer 367 // I til lægen?", "Vinker I til os?") â pronoun, whatever the verb. 368 // Accepted rare miss: a fronted plural noun + preposition ("Regninger 369 // I postkassenâ¦") keeps its capital â cosmetic. 370 // The first word must LOOK verbal (present-tense -r or the finite list) 371 // â a fronted noun ("Mad I ovnen", "Nøglen I døren") is a preposition 372 // context, not inversion (hostile r5). Accepted rest-miss: plural nouns 373 // in -er ("Biler I garagen") read as verbal â cosmetic. 374 if (allowFirstWordInversion && prevMatch && /r$/.test(prevMatch[1])) { 375 const beforePrev = before.slice(0, before.lastIndexOf(prevMatch[1])); 376 if (/(?:^|[.!?â¦]\s*)$/.test(beforePrev)) return false; 377 } 378 if (STT_DA_I_LOWER_WEAK_NEXT.has(key)) return true; 379 return !STT_DA_I_KEEP_NEXT.has(key); 380 }; 381 382 // The next word is a LOOKAHEAD, not consumed â otherwise a second capital I 383 // straight after a kept continuation word could never match ("Som I nævnte 384 // I voresâ¦", hostile r2 mixed case). 385 // [^\S\n] not \s in the lead (hostile r9): a newline IS a sentence 386 // boundary here, and \s would let one through as the separator. 387 text = text.replace( 388 /([^.!?â¦\n][^\S\n])I(?=[^\S\n]+([\p{L}\d][\p{L}\d'â\-]*))/gu, 389 (full, lead, next, off, whole) => 390 lowerCapitalI(next, whole.slice(0, off + lead.length), true) 391 ? lead + 'i' : full 392 ); 393 // Abbreviation dots are not sentence boundaries (hostile r4): "Vi ses 394 // kl. I morgen" â the interior pass excludes '.' in its lead, so known 395 // abbreviations get their own pass with the SAME decision. No first-word 396 // inversion here: the word before the dot is mid-sentence by definition. 397 text = text.replace( 398 /(\b(?:[Kk]l|[Cc]a|[Nn]r|[Dd])\.[^\S\n])I(?=[^\S\n]+([\p{L}\d][\p{L}\d'â\-]*))/gu, 399 (full, lead, next, off, whole) => 400 lowerCapitalI(next, whole.slice(0, off + lead.length), false) 401 ? lead + 'i' : full 402 ); 403 404 // R5 â months are lowercase in Danish, but ONLY in unambiguous contexts. 405 // A digit context ("den 28. August 2026") is always the month â every month 406 // name qualifies. A preposition context ("til August") is ambiguous for the 407 // month names that are also common given names (April, Maj, August, Juni: 408 // "Giv bogen til August", "besøg fra Maj" â lens-found 12/7, semantic harm), 409 // so the preposition trigger only fires for the never-a-name months. 410 // A capitalized word straight after the month is a proper-name context 411 // ("5. Juni Plads" â a real Copenhagen square, hostile r9): skip. 412 text = text.replace( 413 /(\d\.?\s+)(Januar|Februar|Marts|April|Maj|Juni|Juli|August|September|Oktober|November|December)\b(?!\s+[A-ZÃÃà ])/g, 414 (full, lead, month) => lead + month.toLocaleLowerCase('da-DK') 415 ); 416 // Case-tolerant preposition (hostile r1: sentence-initial "I Januarâ¦" / 417 // "Til Januarâ¦" is still the month for the never-a-name set); the 418 // preposition's own casing is preserved via the capture. 419 text = text.replace( 420 /(\b(?:[Ii]|[Tt]il|[Ff]ra|[Ii]nden|[Ss]iden|[Pp]rimo|[Mm]edio|[Uu]ltimo)\s+)(Januar|Februar|Marts|Juli|September|Oktober|November|December)\b/g, 421 (full, lead, month) => lead + month.toLocaleLowerCase('da-DK') 422 ); 423 // Lowercase "i + month" is temporal for EVERY month (hostile r3): Danish 424 // never puts the bare preposition "i" in front of a person's name. Only 425 // the LOWERCASE i triggers for the name-collision months (hostile r6: a 426 // capital "I Maj â¦" can be the pronoun addressing a person named Maj) â 427 // R4 runs first and has already lowered genuine preposition-I before a 428 // month, so "Vi ses I Maj" still ends as "i maj" through the chain.
429 text = text.replace( 430 /(\bi\s+)(April|Maj|Juni|August)\b/g, 431 (full, lead, month) => lead + month.toLocaleLowerCase('da-DK') 432 ); 433 // Sentence-initial "I Maj skalâ¦" is the temporal reading (hostile r10 434 // reversed the r6 guard: a vocative without comma punctuation is not 435 // plausible Danish, and the missing subject is exactly the temporal 436 // clue). The capital I stays â it starts the sentence. 437 text = text.replace( 438 /(^|[.!?â¦]\s+)(I[^\S\n]+)(April|Maj|Juni|August)\b/g, 439 (full, boundary, lead, month) => boundary + lead + month.toLocaleLowerCase('da-DK') 440 ); 441 442 // R6 â the provider capitalizes the word after ANY period, including 443 // abbreviation dots. Tight whitelists only: time words after "kl." and 444 // location/ordinal nouns after a digit ordinal ("3. Sal" â "3. sal"). 445 text = text.replace(/(\b[Kk]l\.\s+)(Halv|Kvart)\b/g, 446 (full, lead, word) => lead + word.toLocaleLowerCase('da-DK')); 447 text = text.replace(/(\d+\.\s+)(Sal|Etage|Klasse|Plads|Række|Afdeling)\b/g, 448 (full, lead, word) => lead + word.toLocaleLowerCase('da-DK')); 449 450 return text; 451 } 452 453 // ââ Multi-language STT post-format registry (#5738 follow-up) ââââââââââââââââ 454 // The STT provider's smart-format layer is English-first, so every non-English 455 // UI language receives English number/date habits. A voice-box probe 456 // (tools/stt-multilang-probe.mjs) proved the per-language quirks â see 457 // docs/stt-language-normalizers.md for the map and the "add a language" recipe. 458 // Adding a language = write a normalizer, register it below, add fixtures, run 459 // the probe. Every rule stays orthography-only: never change digit VALUES, 460 // names, or negations. German is the proof this MUST be per-language: German 461 // months are correctly capitalized, so the Danish month-lowercase rule would 462 // corrupt German â a single shared rule is impossible. 463 464 // Shared primitive: decimal point -> comma with the numeric guards proven on 465 // Danish (a valid HH.MM clock and dotted sequences are preserved; a unit word 466 // forces conversion). Language-agnostic â the caller passes its own unit set 467 // plus optional guards (native lens 13/7): `idRe` protects identifier 468 // prefixes ("Version 2.5", "Punkt 3.2" â semantic labels, never decimals), 469 // `dateRe` protects single-digit date shapes behind that language's date 470 // words ("Am 1.1", "Den 3.5"). Danish keeps its own battle-tested inline copy. 471 function sttDecimalComma(text, unitRe, opts) { 472 const idRe = opts && opts.idRe; 473 const dateRe = opts && opts.dateRe; 474 // Single-digit fraction ("2,5" grades/weights) converts by default; an 475 // identifier prefix always wins, and a 1-31.1-9 date shape survives when 476 // the language's date words precede it. 477 text = text.replace(/(^|[^\d.-])(\d+)\.(\d)(?!\d)(?!\.\d)/g, 478 (full, lead, int, frac, off, whole) => { 479 const before = whole.slice(0, off + lead.length); 480 if (idRe && idRe.test(before)) return full; 481 if (unitRe.test(whole.slice(off + full.length))) return lead + int + ',' + frac; 482 const intNum = parseInt(int, 10); 483 if (dateRe && intNum >= 1 && intNum <= 31 && frac !== '0' 484 && dateRe.test(before)) return full; 485 return lead + int + ',' + frac; 486 }); 487 // Two-digit fraction: a valid clock (HH.MM) or DD.MM/YYYY.MM date is kept 488 // unless a unit word follows; everything else is a decimal. 489 text = text.replace(/(^|[^\d.-])(\d+)\.(\d{2})(?!\d)(?!\.\d)/g, 490 (full, lead, int, frac, off, whole) => { 491 if (idRe && idRe.test(whole.slice(0, off + lead.length))) return full; 492 if (unitRe.test(whole.slice(off + full.length))) return lead + int + ',' + frac; 493 const intNum = parseInt(int, 10); 494 const fracNum = parseInt(frac, 10); 495 if (intNum <= 23 && fracNum <= 59) return full; 496 if (fracNum >= 1 && fracNum <= 12 497 && (intNum <= 31 || /^(?:1[89]|2[01])\d{2}$/.test(int))) return full; 498 return lead + int + ',' + frac; 499 }); 500 return text; 501 } 502 503 // Identifier prefixes shared by the Latin-script languages (native lens 13/7: 504 // "Version 2.5"/"iOS 17.4" were corrupted in every language) + per-language 505 // section words. 506 const STT_ID_DE = /(?:^|[^\p{L}])(?:version|punkt|abschnitt|kapitel|paragraph|absatz|ziffer|artikel|ios|android|gpt-?\d*|v)\.?\s*$/iu; 507 const STT_ID_NO = /(?:^|[^\p{L}])(?:versjon|punkt|kapittel|avsnitt|paragraf|artikkel|ios|android|gpt-?\d*|v)\.?\s*$/iu; 508 const STT_ID_SV = /(?:^|[^\p{L}])(?:version|punkt|kapitel|avsnitt|paragraf|artikel|ios|android|gpt-?\d*|v)\.?\s*$/iu; 509 const STT_ID_FR = /(?:^|[^\p{L}])(?:version|article|section|point|chapitre|ios|android|gpt-?\d*|v)\.?\s*$/iu; 510 const STT_ID_ES = /(?:^|[^\p{L}
510])(?:versión|version|artÃculo|punto|apartado|capÃtulo|sección|ios|android|gpt-?\d*|v)\.?\s*$/iu; 511 512 // Date words that precede a DD.M single-digit date in each language 513 // (native lens: "Am 1.1", "Den 3.5" were corrupted). 514 const STT_DATE_DE = /(?:^|[^\p{L}])(?:am|bis|der|den|zum|ab|frist|termin|datum|liefer\p{L}*)\s*$/iu; 515 const STT_DATE_NO = /(?:^|[^\p{L}])(?:den|frist(?:en)?|møte(?:t)?|dato(?:en)?|innen|senest)\s*$/iu; 516 const STT_DATE_SV = /(?:^|[^\p{L}])(?:den|frist(?:en)?|möte(?:t)?|datum(?:et)?|innan|senast)\s*$/iu; 517 518 // Norwegian: Nordic sibling of Danish â comma decimals, DD.MM dates, lowercase 519 // months (mars/mai/desember spellings). No capital-"I" rule (archaic in NO). 520 const STT_NO_UNIT = 521 /^\s?(?:(?:kilo|kg|gram|g|kroner|kr\.?|prosent|meter|m|liter|l|km|cm|mm|ml|dl|Ã¥r|aar|mÃ¥ned|mÃ¥neder|uke|uker|time|timer|minutt|minutter|sekund|sekunder|poeng)(?=$|[^\p{L}\p{N}_])|%)/iu; 522 // Month split mirrors Danish (native lens 13/7): Mars/April/Mai/Juni/August 523 // double as Norwegian given names or planet â they lower ONLY after a digit 524 // ordinal, and never when the NEXT word is capitalized ("til August Hansen", 525 // "5. Juni Plass" stay). The never-a-name months also lower after i/til/fra. 526 const STT_NO_MONTHS_DIGIT = 527 /(\d\.?\s+)(Januar|Februar|Mars|April|Mai|Juni|Juli|August|September|Oktober|November|Desember)\b(?!\s+[A-ZÃÃà ])/g; 528 const STT_NO_MONTHS_PREP = 529 /(\b(?:[Ii]|[Tt]il|[Ff]ra)\s+)(Januar|Februar|Juli|September|Oktober|November|Desember)\b(?!\s+[A-ZÃÃà ])/g; 530 function normalizeSttNorwegian(input) { 531 let t = String(input || ''); 532 if (!t) return t; 533 t = t.replace(/(\d)\s?[ºª](?!\s*[CFcf]\b)\.?/g, '$1.'); 534 t = sttDecimalComma(t, STT_NO_UNIT, { idRe: STT_ID_NO, dateRe: STT_DATE_NO }); 535 t = t.replace(STT_NO_MONTHS_DIGIT, (full, lead, month) => lead + month.toLocaleLowerCase('nb-NO')); 536 t = t.replace(STT_NO_MONTHS_PREP, (full, lead, month) => lead + month.toLocaleLowerCase('nb-NO')); 537 return t; 538 } 539 540 // Swedish: the provider already lowercases Swedish months and uses "28:e" 541 // ordinals, so only the decimal separator needs fixing. 542 const STT_SV_UNIT = 543 /^\s?(?:(?:kilo|kg|gram|g|kronor|kr\.?|procent|meter|m|liter|l|km|cm|mm|ml|dl|Ã¥r|aar|mÃ¥nad|mÃ¥nader|vecka|veckor|timme|timmar|minut|minuter|sekund|sekunder|poäng)(?=$|[^\p{L}\p{N}_])|%)/iu; 544 function normalizeSttSwedish(input) { 545 let t = String(input || ''); 546 if (!t) return t; 547 return sttDecimalComma(t, STT_SV_UNIT, { idRe: STT_ID_SV, dateRe: STT_DATE_SV }); 548 } 549 550 // German (probe r2): decimals come back dotted ("12.99 Euro", grade "1.3"); 551 // clocks arrive as "14 Uhr 30" (never dotted) and months are CORRECTLY 552 // capitalized â so German gets ONLY the decimal-comma primitive, no month 553 // rule, ever. 554 const STT_DE_UNIT = 555 /^\s?(?:(?:kilo|kg|gramm|g|euro|prozent|meter|m|liter|l|km|cm|mm|ml|dl|punkte?|kalorien|jahre?|monate?|wochen?|stunden?|minuten?|sekunden?)(?=$|[^\p{L}\p{N}_])|%)/iu; 556 function normalizeSttGerman(input) { 557 let t = String(input || ''); 558 if (!t) return t; 559 return sttDecimalComma(t, STT_DE_UNIT, { idRe: STT_ID_DE, dateRe: STT_DATE_DE }); 560 } 561 562 // French (probe r2): dotted decimals ("12.99", "14.5") and "50 pour 100" for 563 // per-cent. The percent rewrite REQUIRES a percent-context word before the 564 // number â "un menu pour 100 personnes" is real French ("for 100 people") and 565 // must never be rewritten. Clocks arrive as "14 heures 30" (never dotted). 566 // The apostrophe is EXCLUDED from the unit boundary (native lens 13/7): the 567 // elided article "l'" ("14.30 l'après-midi") is not the litre unit â treating 568 // it as one overrode the clock/date guards. 569 const STT_FR_UNIT = 570 /^\s?(?:(?:kilos?|kg|grammes?|g|euros?|centimes?|mètres?|m|litres?|l|km|cm|mm|ml|dl|points?|calories|ans?|mois|semaines?|heures?|minutes?|secondes?)(?=$|[^\p{L}\p{N}_'â])|%)/iu; 571 const STT_FR_PERCENT_CONTEXT = 572 /(?:remise|réduction|rabais|taux|augmentation|baisse|hausse|tva|intérêts?|croissance|inflation|bénéfice|chômage|commission|marge|rendement|probabilité|réussite|score)[^\p{L}\n]+(?:[\p{L}'â\-]+[^\p{L}\n]+){0,3}$/iu; 573 function normalizeSttFrench(input) { 574 let t = String(input || ''); 575 if (!t) return t; 576 t = sttDecimalComma(t, STT_FR_UNIT, { idRe: STT_ID_FR }); 577 t = t.replace(/(\d+(?:,\d+)?)\s+pour 100\b/g, (full, num, off) => 578 STT_FR_PERCENT_CONTEXT.test(t.slice(0, off)) ? num + ' pour cent' : full); 579 return t; 580 } 581 582 // Spanish (probe r2): decimals come back as literal "12 coma 99" â the join 583 // requires a DIGIT before "coma", which naturally protects the medical sense 584 // ("Estuvo en coma 5 dÃas" has no digit before it, probe-verified). Percent 585 // arrives as "50 por 100" with the same context requirement as French. 586 const STT_ES_UNIT = 587 /^\s?(?:(?:kilos?|kg|gramos?|g|euros?|céntimos?|metros?|m|litros?|l|km|cm|mm|ml|dl|puntos?|calorÃas|años?|meses|semanas?|horas?|minutos?|segundos?)(?=$|[^\p{L}\p{N}_])|%)/iu;
588 const STT_ES_PERCENT_CONTEXT = 589 /(?:descuento|rebaja|interés|intereses|iva|aumento|subida|bajada|tasa|crecimiento|inflación|margen|ganancia|beneficio|rentabilidad|comisión|desempleo)[^\p{L}\n]+(?:[\p{L}'â\-]+[^\p{L}\n]+){0,3}$/iu; 590 // Enumeration guards for the coma-join (native lens 13/7): "artÃculos 1 coma 591 // 2 del reglamento" is a LIST ("articles 1, 2"), not a decimal â a plural 592 // section noun before the number, or a "y N" continuation after, blocks the 593 // join. Documented rest-miss: an enumeration with neither signal joins. 594 const STT_ES_ENUM_BEFORE = 595 /(?:artÃculos|puntos|apartados|párrafos|ejercicios|apartamentos|números|capÃtulos|páginas|secciones|habitaciones)\s+$/iu; 596 function normalizeSttSpanish(input) { 597 let t = String(input || ''); 598 if (!t) return t; 599 t = t.replace(/(\d+)\s+coma\s+(\d+)/gi, (full, a, b, off, whole) => { 600 if (STT_ES_ENUM_BEFORE.test(whole.slice(0, off))) return full; 601 if (/^\s+y\s+\d/.test(whole.slice(off + full.length))) return full; 602 return a + ',' + b; 603 }); 604 t = sttDecimalComma(t, STT_ES_UNIT, { idRe: STT_ID_ES }); 605 t = t.replace(/(\d+(?:,\d+)?)\s+por 100\b/g, (full, num, off) => 606 STT_ES_PERCENT_CONTEXT.test(t.slice(0, off)) ? num + ' por ciento' : full); 607 return t; 608 } 609 610 // The registry. Adding a language = one entry + a fixture. Deliberately 611 // absent: English (its point + capitalization are already correct), Ukrainian 612 // (probe r2: the provider's uk output mixes words and digits mid-number â 613 // "дванадÑÑÑÑ, 99" â no deterministic rule can join that safely; clocks 614 // "14:30" and "50%" already arrive correct), Arabic (needs an explicit 615 // product decision on Arabic-Indic digits + the Ù« decimal sign + RTL before 616 // any rule ships). See docs/stt-language-normalizers.md. 617 const STT_LANG_NORMALIZERS = { 618 da: normalizeSttDanishText, 619 no: normalizeSttNorwegian, 620 sv: normalizeSttSwedish, 621 de: normalizeSttGerman, 622 fr: normalizeSttFrench, 623 es: normalizeSttSpanish, 624 }; 625 626 // One dispatch for the whole rail: every mic passes its resolved short code. 627 // An unregistered language (en and the not-yet-hardened ones) is a passthrough. 628 function normalizeSttText(text, lang) { 629 const fn = STT_LANG_NORMALIZERS[lang]; 630 return fn ? fn(text) : String(text || ''); 631 } 632 633 const TARGET_SAMPLE_RATE = 16000; 634 635 /** 636 * Stateful linear-interpolation resampler for mono Int16 PCM chunks. 637 * The voice websocket expects 16 kHz; browsers that refuse a fixed-rate 638 * AudioContext (Firefox throws NotSupportedError when the context rate 639 * differs from the microphone's native rate) capture at native rate and 640 * convert here instead. Keeps the fractional read position and the previous 641 * chunk's last sample so chunk boundaries stay continuous. 642 * Returns null when no conversion is needed. 643 */ 644 function createPcm16Resampler(fromRate, toRate) { 645 if (!fromRate || !toRate || fromRate === toRate) return null; 646 const step = fromRate / toRate; 647 let pos = 0; 648 let tail = null; 649 650 return function resample(input) { 651 const prefix = tail === null ? 0 : 1; 652 const total = input.length + prefix; 653 if (total < 2) { 654 if (input.length) tail = input[input.length - 1]; 655 return new Int16Array(0); 656 } 657 const sampleAt = (i) => (prefix && i === 0) ? tail : input[i - prefix]; 658 const last = total - 1; 659 const count = pos > last ? 0 : Math.floor((last - pos) / step) + 1; 660 const out = new Int16Array(count); 661 for (let n = 0; n < count; n++) { 662 const base = Math.floor(pos); 663 const frac = pos - base; 664 const a = sampleAt(base); 665 const b = base < last ? sampleAt(base + 1) : a; 666 out[n] = Math.round(a + (b - a) * frac); 667 pos += step; 668 } 669 pos -= last; 670 tail = sampleAt(last); 671 return out; 672 }; 673 } 674 675 function createStt$1(config) { 676 const MAX_RECONNECT = 3; 677 const RECONNECT_BASE_MS = 1000; 678 const WS_CONNECT_TIMEOUT_MS = 8000; 679 const TOKEN_TIMEOUT_MS = 8000; 680 const SEND_BATCH_SAMPLES = 1024; 681 const PCM_WORKLET_URL = '/shared/js/pcm-worklet.js'; 682 const suppressToast = config.suppressToast || false; 683 684 // undefined = UI-language mic (resolved fresh per token fetch, so a language 685 // switch mid-page is honored without recreating the instance); string = fixed 686 // content language; null = provider auto-detect. setLang() pins it explicitly. 687 let lang = config.lang; 688 const service = config.service || 'chat'; 689 const tokenEndpoint = config.tokenEndpoint 690 || (typeof window !== 'undefined' && window.__SS_STT_TOKEN_ENDPOINT) 691 || '/api/stt-token.php'; 692 const onTranscript = config.onTranscript; 693 const onPartial = config.onPartial || null; 694 const onAudioActivity = config.onAudioActivity || null; 695 const onError = config.onError || null; 696 const onStateChange = config.onStateChange || null; 697 const btnEl = config.button || null; 698 699 function emitError(msg, type) { 700 if (!suppressToast) _showSttToast$1(msg); 701 if (onError) onError(msg, type || 'unknown'); 702 } 703 const idleAriaLabel = config.idleAriaLabel || (btnEl ? (btnEl.getAttribute('aria-label') || '') : ''); 704 const ariaLabels = config.ariaLabels || {}; 705 706 const SILENCE_TIMEOUT_MS = 30000; 707 let state = 'idle'; 708 let audioCtx = null; 709 let workletNode = null; 710 let sourceNode = null; 711 let stream = null; 712 let socket = null; 713 let silenceTimer = null; 714 let reconnectCount = 0; 715 let reconnectTimer = null; 716 let sessionId = 0; 717 let destroyed = false;
718 let lastSnippet = ''; 719 // Dedup window (hostile r10+r12): provider duplicate finals arrive within 720 // milliseconds; a genuine fast repeated utterance ("ja" ⦠"ja") takes 721 // longer than 400 ms and must deliver. 722 let lastSnippetAt = 0; 723 const voiceActivity = onAudioActivity 724 ? createVoiceActivityDetector(onAudioActivity) 725 : null; 726 const FINAL_DEDUP_WINDOW_MS = 400; 727 728 function setButtonState(s) { 729 if (!btnEl) return; 730 731 btnEl.classList.add('ss-mic'); 732 btnEl.classList.remove( 733 'ss-mic--connecting', 734 'ss-mic--listening', 735 'ss-mic--recording', 736 'ss-mic--error', 737 'ss-mic--processing', 738 'ss-mic--done' 739 ); 740 if (s === 'connecting') { 741 btnEl.classList.add('ss-mic--connecting'); 742 } else if (s === 'listening') { 743 btnEl.classList.add('ss-mic--listening'); 744 } else if (s === 'error') { 745 btnEl.classList.add('ss-mic--error'); 746 } 747 748 const running = s === 'listening' || s === 'connecting'; 749 btnEl.setAttribute('aria-busy', s === 'connecting' ? 'true' : 'false'); 750 btnEl.setAttribute('aria-pressed', running ? 'true' : 'false'); 751 752 const label = s === 'connecting' 753 ? (ariaLabels.connecting || 'Forbinder...') 754 : s === 'listening' 755 ? (ariaLabels.listening || 'Lytter...') 756 : s === 'error' 757 ? (ariaLabels.error || idleAriaLabel) 758 : idleAriaLabel; 759 if (label) btnEl.setAttribute('aria-label', label); 760 } 761 762 function resetSilenceTimer() { 763 clearTimeout(silenceTimer); 764 if (state === 'listening') { 765 silenceTimer = setTimeout(() => { 766 if (state === 'listening') { 767 emitError('Ingen tale registreret â prøv igen', 'silence'); 768 stop(); 769 } 770 }, SILENCE_TIMEOUT_MS); 771 } 772 } 773 774 function setState(s) { 775 if (state === s) return; 776 state = s; 777 setButtonState(s); 778 if (s === 'listening') resetSilenceTimer(); 779 if (s === 'idle' || s === 'error') clearTimeout(silenceTimer); 780 if (onStateChange) onStateChange(s); 781 } 782 783 function cleanup() { 784 if (reconnectTimer) { clearTimeout(reconnectTimer); reconnectTimer = null; } 785 if (workletNode) { try { workletNode.disconnect(); } catch {} workletNode = null; } 786 if (sourceNode) { try { sourceNode.disconnect(); } catch {} sourceNode = null; } 787 if (stream) { stream.getTracks().forEach(t => t.stop()); stream = null; } 788 if (audioCtx && audioCtx.state !== 'closed') { try { audioCtx.close(); } catch {} } 789 audioCtx = null; 790 if (socket) { try { if (socket.readyState <= 1) socket.close(); } catch {} socket = null; } 791 } 792 793 function isCurrentSession(id) { 794 return !destroyed && id === sessionId; 795 } 796 797 function scheduleReconnect(id) { 798 if (!isCurrentSession(id) || state !== 'connecting') return; 799 800 if (reconnectCount < MAX_RECONNECT) { 801 reconnectCount++; 802 const delay = RECONNECT_BASE_MS * reconnectCount; 803 reconnectTimer = setTimeout(() => { 804 reconnectTimer = null; 805 if (isCurrentSession(id) && state === 'connecting') startAttempt(id); 806 }, delay); 807 return; 808 } 809 810 reconnectCount = 0; 811 emitError('Forbindelse til taletjeneste fejlede â prøv igen', 'service'); 812 setState('error'); 813 setTimeout(() => { if (isCurrentSession(id)) setState('idle'); }, 1500); 814 } 815 816 function createTokenTimeoutSignal(ms) { 817 if (typeof AbortSignal !== 'undefined' && typeof AbortSignal.timeout === 'function') { 818 return { signal: AbortSignal.timeout(ms), cleanup: null }; 819 } 820 if (typeof AbortController !== 'undefined') { 821 const controller = new AbortController(); 822 const timer = setTimeout(() => controller.abort(), ms); 823 return { signal: controller.signal, cleanup: () => clearTimeout(timer) }; 824 } 825 return { signal: null, cleanup: null }; 826 } 827 828 async function fetchToken() { 829 const isSkrivSikkertStt = /\/SkrivSikkertSTT-token(?:[/?#]|$)/.test(String(tokenEndpoint)); 830 const effectiveLang = isSkrivSikkertStt 831 ? null
832 : (lang === undefined ? resolveSttUiLang() : lang); 833 // SkrivSikkertSTT uses Google language detection, so its request contains no 834 // lang value at all. Legacy token services retain their existing language. 835 const attemptLang = _sttShortLang(effectiveLang); 836 const langParam = effectiveLang ? '&lang=' + encodeURIComponent(effectiveLang) : ''; 837 // sp = page path: the server maps it through its surface-allowlist for the 838 // usage ledger (Referer strippes af privacy-indstillinger/visse mobil- 839 // browsere, som fejl-arkiverede disse kald som surface=unknown). 840 const spParam = (typeof location !== 'undefined' && location.pathname) 841 ? '&sp=' + encodeURIComponent(location.pathname) : ''; 842 const separator = String(tokenEndpoint).includes('?') ? '&' : '?'; 843 const query = 'service=' + encodeURIComponent(service) + langParam + spParam; 844 const tokenUrl = String(tokenEndpoint) + separator + query; 845 846 const opts = {}; 847 let sameOriginTokenEndpoint = true; 848 try { 849 sameOriginTokenEndpoint = new URL(tokenUrl, location.href).origin === location.origin; 850 } catch (e) { /* keep same-origin compatibility in restricted runtimes */ } 851 // Identified token-fetch: the server can only issue per-user keyterms (the 852 // dictation dictionary) when it knows who is asking. Fail-quiet â with no 853 // client/session the call behaves exactly like the anonymous call always has. 854 if (sameOriginTokenEndpoint) { 855 try { 856 const sb = (typeof window !== 'undefined') ? window._supabase : null; 857 if (sb && sb.auth && typeof sb.auth.getSession === 'function') { 858 // Cap the session read: if Supabase JS is mid-refresh or blocked on 859 // storage, mic-open must not hang before the token timeout even starts. 860 const r = await Promise.race([ 861 sb.auth.getSession(), 862 new Promise((res) => setTimeout(() => res(null), 800)), 863 ]); 864 const at = r && r.data && r.data.session && r.data.session.access_token; 865 if (at) opts.headers = { Authorization: 'Bearer ' + at }; 866 } 867 } catch (e) { /* anonymous fallback */ } 868 } 869 const timeout = createTokenTimeoutSignal(TOKEN_TIMEOUT_MS); 870 if (timeout.signal) opts.signal = timeout.signal; 871 let resp; 872 try { 873 resp = await fetch(tokenUrl, opts); 874 } finally { 875 if (timeout.cleanup) timeout.cleanup(); 876 } 877 if (!resp.ok && opts.headers && (resp.status === 401 || resp.status === 403)) { 878 // ONLY an auth failure (expired/invalid session) retries anonymously â 879 // not every 4xx â so a 400/404/429 is not silently duplicated. The mic 880 // must never die because the user's session lapsed. 881 const t2 = createTokenTimeoutSignal(TOKEN_TIMEOUT_MS); 882 const o2 = {}; 883 if (t2.signal) o2.signal = t2.signal; 884 try { 885 resp = await fetch(tokenUrl, o2); 886 } finally { 887 if (t2.cleanup) t2.cleanup(); 888 } 889 } 890 if (!resp.ok) { 891 const err = new Error('Token fetch failed: ' + resp.status); 892 // 4xx is deterministic (auth/validation) â retrying cannot fix it. 893 err.sttPermanent = resp.status >= 400 && resp.status < 500; 894 throw err; 895 } 896 const data = await resp.json(); 897 if (!data || typeof data.url !== 'string' || data.url.trim() === '') { 898 const err = new Error('Token response invalid'); 899 err.sttPermanent = true; 900 throw err; 901 } 902 return { url: data.url.trim(), lang: attemptLang }; 903 } 904 905 async function connectSocket(wsUrl, id, attemptLang) { 906 // The Danish-format decision travels WITH this attempt (hostile r5+r10): 907 // a late message on a still-draining socket must never be formatted by 908 // a newer token fetch's language gate â no shared mutable state at all. 909 return new Promise((resolve, reject) => { 910 const ws = new WebSocket(wsUrl); 911 let settled = false; 912 913 // A blackholed wss endpoint (proxy/firewall) never fires open/error; 914 // without a deadline the button would spin forever. 915 const connectTimer = setTimeout(() => { 916 fail(new Error('WebSocket connect timeout')); 917 try { ws.close(); } catch {} 918 }, WS_CONNECT_TIMEOUT_MS); 919 920 function fail(err) { 921 if (settled) return; 922 settled = true; 923 clearTimeout(connectTimer); 924 reject(err); 925 } 926 927 ws.onopen = () => { 928 socket = ws; 929 }; 930 931 ws.onmessage = (evt) => { 932 try { 933 const msg = JSON.parse(evt.data); 934 switch (msg.type) { 935 case 'ready': 936 if (!isCurrentSession(id)) { 937 try { ws.close(); } catch {} 938 fail(new Error('WebSocket cancelled')); 939 return; 940 } 941 if (settled) return; 942 settled = true; 943 clearTimeout(connectTimer); 944 resolve(ws); 945 break; 946 case 'partial': 947 if (msg.text && onPartial) { 948 resetSilenceTimer(); 949 onPartial(normalizeSttText(msg.text, attemptLang)); 950 } 951 break; 952 case 'final': { 953 if (!msg.text) break; 954 // Dedup on the EMITTED text (hostile r5 â conceded): two raw 955 // finals differing only in normalizable form are the same 956 // utterance;
956 the user-visible transcript is what must not 957 // double-insert. 958 const transcriptText = normalizeSttText(msg.text, attemptLang); 959 const now = Date.now(); 960 // Window-scoped dedup (hostile r10): only back-to-back 961 // duplicates are the provider double-delivery; a genuinely 962 // repeated utterance later must still come through. 963 if (transcriptText !== lastSnippet 964 || now - lastSnippetAt > FINAL_DEDUP_WINDOW_MS) { 965 lastSnippet = transcriptText; 966 lastSnippetAt = now; 967 resetSilenceTimer(); 968 if (onTranscript) onTranscript(transcriptText); 969 } 970 break; 971 } 972 case 'error': 973 emitError('Taletjeneste fejl â prøv igen', 'service'); 974 if (!settled) fail(new Error(msg.message || 'WebSocket error')); 975 break; 976 } 977 } catch {} 978 }; 979 980 ws.onerror = () => fail(new Error('WebSocket error')); 981 982 ws.onclose = () => { 983 if (!settled) { 984 fail(new Error('WebSocket closed before ready')); 985 return; 986 } 987 if (state === 'listening' && !destroyed) { 988 cleanup(); 989 if (isCurrentSession(id)) { 990 setState('connecting'); 991 scheduleReconnect(id); 992 } 993 } 994 }; 995 }); 996 } 997 998 async function setupAudio(ws) { 999 const AudioCtx = window.AudioContext || window.webkitAudioContext; 1000 try { 1001 audioCtx = new AudioCtx({ sampleRate: TARGET_SAMPLE_RATE }); 1002 } catch { 1003 // Some browsers refuse a forced context rate entirely. 1004 audioCtx = new AudioCtx(); 1005 } 1006 if (audioCtx.state === 'suspended') await audioCtx.resume(); 1007 1008 try { 1009 sourceNode = audioCtx.createMediaStreamSource(stream); 1010 } catch (err) { 1011 // Firefox cannot connect a mic stream to a context running at a 1012 // different rate than the device (NotSupportedError). Recreate the 1013 // context at the device's native rate; the resampler below converts 1014 // to the 16 kHz the voice websocket expects. 1015 if (!err || err.name !== 'NotSupportedError') throw err; 1016 try { audioCtx.close(); } catch {} 1017 audioCtx = new AudioCtx(); 1018 if (audioCtx.state === 'suspended') await audioCtx.resume(); 1019 sourceNode = audioCtx.createMediaStreamSource(stream); 1020 } 1021 1022 const resample = createPcm16Resampler(audioCtx.sampleRate, TARGET_SAMPLE_RATE); 1023 let pending = []; 1024 let pendingSamples = 0; 1025 const sendInt16 = (int16) => { 1026 if (!int16 || int16.length === 0) return; 1027 // Native-rate capture fires more, smaller chunks than the 16 kHz path; 1028 // batch to ~64 ms so the socket is not flooded with tiny frames. 1029 pending.push(int16); 1030 pendingSamples += int16.length; 1031 if (pendingSamples < SEND_BATCH_SAMPLES) return; 1032 const merged = new Int16Array(pendingSamples); 1033 let offset = 0; 1034 for (const chunk of pending) { merged.set(chunk, offset); offset += chunk.length; } 1035 pending = []; 1036 pendingSamples = 0; 1037 if (ws.readyState === WebSocket.OPEN) ws.send(merged.buffer); 1038 }; 1039 1040 if (typeof audioCtx.audioWorklet?.addModule === 'function') { 1041 await audioCtx.audioWorklet.addModule(PCM_WORKLET_URL); 1042 workletNode = new AudioWorkletNode(audioCtx, 'pcm-capture'); 1043 workletNode.port.onmessage = (e) => { 1044 if (ws.readyState !== WebSocket.OPEN) return; 1045 // Preserve the original zero-copy path for every ordinary STT caller. 1046 // Only talk-mode opts into local voice activity; resampling still needs 1047 // an Int16 view exactly as it did before this optional hook existed. 1048 if (!voiceActivity && !resample) { ws.send(e.data); return; } 1049 const int16 = new Int16Array(e.data); 1050 if (voiceActivity) voiceActivity.process(int16); 1051 if (!resample) { ws.send(e.data); return; } 1052 sendInt16(resample(int16)); 1053 }; 1054 sourceNode.connect(workletNode); 1055 workletNode.connect(audioCtx.destination); 1056 } else { 1057 const processor = audioCtx.createScriptProcessor(4096, 1, 1); 1058 processor.onaudioprocess = (e) => { 1059 if (ws.readyState !== WebSocket.OPEN) return; 1060 const float32 = e.inputBuffer.getChannelData(0);
1061 const int16 = new Int16Array(float32.length); 1062 for (let i = 0; i < float32.length; i++) { 1063 const s = Math.max(-1, Math.min(1, float32[i])); 1064 int16[i] = s < 0 ? s * 32768 : s * 32767; 1065 } 1066 if (voiceActivity) voiceActivity.process(int16); 1067 if (!resample) { ws.send(int16.buffer); return; } 1068 sendInt16(resample(int16)); 1069 }; 1070 sourceNode.connect(processor); 1071 processor.connect(audioCtx.destination); 1072 workletNode = processor; 1073 } 1074 } 1075 1076 async function startAttempt(id) { 1077 setState('connecting'); 1078 1079 try { 1080 stream = await navigator.mediaDevices.getUserMedia({ 1081 audio: { channelCount: 1, sampleRate: 16000, echoCancellation: true, noiseSuppression: true } 1082 }); 1083 if (!isCurrentSession(id) || state !== 'connecting') { 1084 cleanup(); 1085 return; 1086 } 1087 } catch (err) { 1088 cleanup(); 1089 // Session abandoned mid-connect (e.g. the user clicked generate / "Skriv 1090 // teksten", which calls stop() and bumps sessionId) â don't emit a late 1091 // permission error or stamp error state on a screen they've moved on from. 1092 // Mirrors the fetchToken() catch guard below (#5261 follow-up). 1093 if (!isCurrentSession(id) || state !== 'connecting') return; 1094 if (err.name === 'NotAllowedError' || err.name === 'SecurityError') { 1095 emitError('Giv mikrofontilladelse i din browser', 'permission'); 1096 } else if (err.name === 'NotFoundError') { 1097 emitError('Ingen mikrofon fundet', 'no-device'); 1098 } else if (err.name === 'NotReadableError') { 1099 emitError('Mikrofonen er optaget af et andet program', 'busy'); 1100 } else if (err.name === 'OverconstrainedError') { 1101 emitError('Mikrofon understøtter ikke de krævede indstillinger', 'device'); 1102 } else { 1103 emitError('Mikrofon kunne ikke startes', 'device'); 1104 } 1105 setState('error'); 1106 setTimeout(() => { if (isCurrentSession(id)) setState('idle'); }, 1500); 1107 return; 1108 } 1109 1110 let token; 1111 try { 1112 token = await fetchToken(); 1113 } catch (err) { 1114 cleanup(); 1115 if (!isCurrentSession(id) || state !== 'connecting') return; 1116 if (err && err.sttPermanent) { 1117 reconnectCount = 0; 1118 emitError('Forbindelse til taletjeneste fejlede â prøv igen', 'service'); 1119 setState('error'); 1120 setTimeout(() => { if (isCurrentSession(id)) setState('idle'); }, 1500); 1121 return; 1122 } 1123 scheduleReconnect(id); 1124 return; 1125 } 1126 if (!isCurrentSession(id) || state !== 'connecting') { 1127 cleanup(); 1128 return; 1129 } 1130 1131 try { 1132 const ws = await connectSocket(token.url, id, token.lang); 1133 if (!isCurrentSession(id) || state !== 'connecting') { 1134 cleanup(); 1135 return; 1136 } 1137 1138 await setupAudio(ws); 1139 if (!isCurrentSession(id) || state !== 'connecting') { 1140 cleanup(); 1141 return; 1142 } 1143 reconnectCount = 0; 1144 setState('listening'); 1145 } catch { 1146 cleanup(); 1147 if (isCurrentSession(id) && state === 'connecting') scheduleReconnect(id); 1148 } 1149 } 1150 1151 async function start() { 1152 if (state === 'listening' || state === 'connecting' || destroyed) return; 1153 1154 sessionId++; 1155 reconnectCount = 0; 1156 lastSnippet = ''; 1157 if (voiceActivity) voiceActivity.reset(); 1158 if (!navigator.mediaDevices || typeof navigator.mediaDevices.getUserMedia !== 'function') { 1159 // Insecure contexts and legacy browsers have no mediaDevices at all;
1160 // fail with guidance instead of a TypeError swallowed as 'device'. 1161 const id = sessionId; 1162 emitError('Din browser understøtter ikke diktering', 'unsupported'); 1163 setState('error'); 1164 setTimeout(() => { if (isCurrentSession(id)) setState('idle'); }, 1500); 1165 return; 1166 } 1167 await startAttempt(sessionId); 1168 } 1169 1170 function stop() { 1171 if (state === 'idle') return; 1172 sessionId++; 1173 setState('idle'); 1174 cleanup(); 1175 } 1176 1177 function toggle() { 1178 if (state === 'listening' || state === 'connecting') { 1179 stop(); 1180 } else { 1181 start(); 1182 } 1183 } 1184 1185 function setLang(l) { 1186 lang = l; 1187 } 1188 1189 function isActive() { 1190 return state === 'listening' || state === 'connecting'; 1191 } 1192 1193 function destroy() { 1194 destroyed = true; 1195 sessionId++; 1196 cleanup(); 1197 state = 'idle'; 1198 setButtonState('idle'); 1199 } 1200 1201 return { start, stop, toggle, destroy, isActive, setLang }; 1202 } 1203 1204 /** 1205 * Browser-native speech-to-text (Web Speech API / webkitSpeechRecognition) 1206 * with automatic fallback to the WebSocket STT engine in shared/js/stt.js. 1207 * 1208 * This is the ONE shared entry point for every mic on the site: it exposes the 1209 * exact same `createStt(config)` contract as shared/js/stt.js (same config 1210 * keys, same returned controller `{ start, stop, toggle, destroy, isActive, 1211 * setLang }`, same state names 'idle'|'connecting'|'listening'|'error', same 1212 * error types and mic-button/aria handling), so consumers switch engines by 1213 * changing only their import path. 1214 * 1215 * Engine choice per instance: 1216 * - `config.onAudioActivity` present (talk-mode voice-barge needs raw PCM 1217 * for local VAD â the Web Speech API exposes no audio) â WebSocket engine. 1218 * - No SpeechRecognition constructor (Firefox, insecure context) â WebSocket 1219 * engine. 1220 * - A previous fatal Web Speech failure on this page load (sticky flag) 1221 * â WebSocket engine. 1222 * - Otherwise â browser-native recognition, and on a fatal runtime failure 1223 * the controller silently swaps itself to a WebSocket instance mid-session 1224 * (restarting capture if the user was dictating), so the mic never dies. 1225 * 1226 * Recognition language is explicitly fixed to Danish (`da-DK`) so Chrome does 1227 * not infer a language from the page or browser configuration. The WebSocket 1228 * fallback keeps its own unchanged language resolution (UI language / 1229 * explicit code / provider auto-detect). 1230 * 1231 * Permission and device errors never fall back: the WebSocket engine needs 1232 * the same microphone, so a denied/missing mic would only fail twice. 1233 */ 1234 1235 1236 const LOG = '[SS STT webkit]'; 1237 const RECOGNITION_LANGUAGE = 'da-DK'; 1238 // Finals/partials run through the same Danish orthography normalizer the 1239 // WebSocket rail applies (decimal commas, preposition-"i", months â¦) so the 1240 // two engines produce identically formatted text. 1241 const NORMALIZE_LANG = 'da'; 1242 1243 const RAPID_END_MS = 1000; 1244 const MAX_RAPID_FAILURES = 3; 1245 const SILENCE_TIMEOUT_MS = 30000; 1246 const ERROR_SETTLE_MS = 1500; 1247 // Upper bound on the stop() flush: Chrome reliably raises onend after stop(); 1248 // this only keeps the mic UI from wedging if a recognizer hangs. 1249 const STOP_FLUSH_TIMEOUT_MS = 2000; 1250 1251 // Sticky for the page's lifetime: after one fatal Web Speech failure every 1252 // later mic goes straight to the WebSocket engine instead of failing again. 1253 let _webkitBroken = false; 1254 1255 let _toastEl = null; 1256 let _toastTimer = null; 1257 function _showSttToast(msg, duration = 4000) { 1258 if (!_toastEl) { 1259 _toastEl = document.createElement('div'); 1260 _toastEl.className = 'ss-stt-toast'; 1261 _toastEl.setAttribute('role', 'alert'); 1262 _toastEl.setAttribute('aria-live', 'assertive'); 1263 document.body.appendChild(_toastEl); 1264 } 1265 _toastEl.textContent = msg; 1266 _toastEl.classList.add('ss-stt-toast--visible'); 1267 clearTimeout(_toastTimer); 1268 _toastTimer = setTimeout(() => { 1269 _toastEl.classList.remove('ss-stt-toast--visible'); 1270 }, duration); 1271 } 1272 1273 function recognitionConstructor() { 1274 if (typeof window === 'undefined') return null; 1275 return window.SpeechRecognition || window.webkitSpeechRecognition || null; 1276 } 1277 1278 // Android Chrome's speech service double-delivers finals (identical repeats 1279 // and cumulative re-sends) in patterns the Web Speech result stream does not 1280 // let us distinguish reliably, so Android routes to the WebSocket engine â 1281 // the previous STT â for every mic. iOS and desktop keep the browser engine. 1282 function isAndroid() { 1283 try { 1284 return typeof navigator !== 'undefined' && /Android/i.test(navigator.userAgent || ''); 1285 } catch (e) { 1286 return false;
1287 } 1288 } 1289 1290 /** 1291 * Same contract as shared/js/stt.js createStt(config). 1292 * @returns {{ start, stop, toggle, destroy, isActive, setLang }} 1293 */ 1294 /* ââ One microphone at a time ââââââââââââââââââââââââââââââââââââââââââââââ 1295 A page has exactly one microphone, but nothing stopped two controllers from 1296 holding it at once. In the email composer that is visible: the toolbar 1297 dictation button (#email-compose-toolbar-dictate, service 1298 'email-compose-body') and Sofia's mic (.ask-ai-mic, service 'email-compose') 1299 are separate createStt instances, so tapping the second while the first ran 1300 left BOTH recognizers listening and both writing transcripts. 1301 1302 Every controller registers here, and one that is about to start stops the 1303 others first. Same single-slot idiom the other robots already implement by 1304 hand â skrivebord's claimMic(), the translator bridge's stopOther() â lifted 1305 into the shared factory so it holds for every mic on the site, including any 1306 added later. Stopping (never aborting) the other controller means its 1307 in-flight audio still flushes its final transcript. */ 1308 const _liveControllers = new Set(); 1309 1310 function _claimMicrophone(self) { 1311 _liveControllers.forEach((other) => { 1312 if (other === self) return; 1313 try { 1314 if (other.isActive()) other.stop(); 1315 } catch (e) { 1316 /* a controller mid-teardown must never block the one starting */ 1317 } 1318 }); 1319 } 1320 1321 function createWebSocketWithPreview(config) { 1322 // The existing socket engine ends synchronously instead of flushing a last 1323 // final. Keep its old commit-on-idle preview behavior for dictation callers; 1324 // talk-mode VAD owns its own utterance completion and must remain untouched. 1325 if (!config.onPartial || config.onAudioActivity) return createStt$1(config); 1326 let partial = ''; 1327 return createStt$1({ 1328 ...config, 1329 onPartial(text) { 1330 partial = text || ''; 1331 config.onPartial(text); 1332 }, 1333 onTranscript(text) { 1334 partial = ''; 1335 if (config.onTranscript) config.onTranscript(text); 1336 }, 1337 onError(message, type) { 1338 partial = ''; 1339 if (config.onError) config.onError(message, type); 1340 }, 1341 onStateChange(state) { 1342 const tail = state === 'idle' ? partial : ''; 1343 if (state === 'idle' || state === 'error' || state === 'connecting') partial = ''; 1344 if (tail && config.onTranscript) config.onTranscript(tail); 1345 if (config.onStateChange) config.onStateChange(state); 1346 }, 1347 }); 1348 } 1349 1350 function createStt(config) { 1351 const engine = config.onAudioActivity || _webkitBroken || isA
1351ndroid() || !recognitionConstructor() 1352 ? createWebSocketWithPreview(config) 1353 : createHybridStt(config); 1354 let destroyed = false; 1355 // Own the slot above engine selection: Android, talk-mode VAD, and a 1356 // native controller that falls back must obey the same exclusion rule. 1357 const controller = { 1358 start() { 1359 if (destroyed || engine.isActive()) return; 1360 _claimMicrophone(controller); 1361 return engine.start(); 1362 }, 1363 stop() { if (!destroyed) return engine.stop(); }, 1364 toggle() { 1365 if (destroyed) return; 1366 if (!engine.isActive()) _claimMicrophone(controller); 1367 return engine.toggle(); 1368 }, 1369 destroy() { 1370 if (destroyed) return; 1371 destroyed = true; 1372 _liveControllers.delete(controller); 1373 return engine.destroy(); 1374 }, 1375 isActive() { return !destroyed && engine.isActive(); }, 1376 setLang(lang) { if (!destroyed) return engine.setLang(lang); }, 1377 }; 1378 _liveControllers.add(controller); 1379 return controller; 1380 } 1381 1382 function createHybridStt(config) { 1383 const suppressToast = config.suppressToast || false; 1384 const onTranscript = config.onTranscript; 1385 const onPartial = config.onPartial || null; 1386 const onError = config.onError || null; 1387 const onStateChange = config.onStateChange || null; 1388 const btnEl = config.button || null; 1389 const idleAriaLabel = config.idleAriaLabel || (btnEl ? (btnEl.getAttribute('aria-label') || '') : ''); 1390 const ariaLabels = config.ariaLabels || {}; 1391 1392 // setLang() must survive a mid-session engine swap: the WebSocket instance 1393 // created on fallback is handed the latest language, not the construction- 1394 // time one (the Web Speech engine itself is fixed to Danish). 1395 let lang = config.lang; 1396 1397 let legacy = null; // WebSocket engine instance after fallback â all methods delegate once set 1398 let state = 'idle'; 1399 let destroyed = false; 1400 let recognition = null; 1401 let active = false; 1402 let shouldRestart = false;
1403 let startedAt = 0; 1404 let rapidFailures = 0; 1405 let sessionId = 0; 1406 let silenceTimer = null; 1407 let settleTimer = null; 1408 // stop() flush window (extension parity): stop() asks Chrome to finalize 1409 // audio it already captured, and those last finals must still reach 1410 // onTranscript while the session winds down â abort() would discard the 1411 // user's final words. 1412 let stoppingFlush = false; 1413 let stopFlushTimer = null; 1414 1415 function emitError(msg, type) { 1416 if (!suppressToast) _showSttToast(msg); 1417 if (onError) onError(msg, type || 'unknown'); 1418 } 1419 1420 // Identical presentation to shared/js/stt.js so a mic button looks and 1421 // announces the same whichever engine drives it. 1422 function setButtonState(s) { 1423 if (!btnEl) return; 1424 1425 btnEl.classList.add('ss-mic'); 1426 btnEl.classList.remove( 1427 'ss-mic--connecting', 1428 'ss-mic--listening', 1429 'ss-mic--recording', 1430 'ss-mic--error', 1431 'ss-mic--processing', 1432 'ss-mic--done' 1433 ); 1434 if (s === 'connecting') { 1435 btnEl.classList.add('ss-mic--connecting'); 1436 } else if (s === 'listening') { 1437 btnEl.classList.add('ss-mic--listening'); 1438 } else if (s === 'error') { 1439 btnEl.classList.add('ss-mic--error'); 1440 } 1441 1442 const running = s === 'listening' || s === 'connecting'; 1443 btnEl.setAttribute('aria-busy', s === 'connecting' ? 'true' : 'false'); 1444 btnEl.setAttribute('aria-pressed', running ? 'true' : 'false'); 1445 1446 const label = s === 'connecting' 1447 ? (ariaLabels.connecting || 'Forbinder...') 1448 : s === 'listening' 1449 ? (ariaLabels.listening || 'Lytter...') 1450 : s === 'error' 1451 ? (ariaLabels.error || idleAriaLabel) 1452 : idleAriaLabel; 1453 if (label) btnEl.setAttribute('aria-label', label); 1454 } 1455 1456 function resetSilenceTimer() { 1457 clearTimeout(silenceTimer); 1458 if (state === 'listening') { 1459 silenceTimer = setTimeout(() => { 1460 if (state === 'listening') { 1461 emitError('Ingen tale registreret â prøv igen', 'silence'); 1462 stopWebkit(); 1463 } 1464 }, SILENCE_TIMEOUT_MS); 1465 } 1466 } 1467 1468 function setState(s) { 1469 if (state === s) return; 1470 state = s; 1471 setButtonState(s); 1472 if (s === 'listening') resetSilenceTimer(); 1473 if (s === 'idle' || s === 'error') clearTimeout(silenceTimer); 1474 if (onStateChange) onStateChange(s); 1475 } 1476 1477 function isCurrent(rec, id) { 1478 return !destroyed && !legacy && recognition === rec && id === sessionId; 1479 } 1480 1481 function teardownRecognition() { 1482 clearTimeout(silenceTimer); 1483 clearTimeout(settleTimer); 1484 clearTimeout(stopFlushTimer); 1485 stopFlushTimer = null; 1486 stoppingFlush = false; 1487 const rec = recognition; 1488 recognition = null; 1489 sessionId++; 1490 if (rec) { 1491 try { 1492 if (typeof rec.abort === 'function') rec.abort(); 1493 else rec.stop(); 1494 } catch (err) { 1495 console.warn(LOG, 'abort failed:', err); 1496 } 1497 } 1498 } 1499 1500 // Error that ends the session on THIS engine without trying the other one 1501 // (permission / missing device â the WebSocket engine shares the mic). 1502 function failLocal(msg, type) { 1503 active = false; 1504 shouldRestart = false;
1505 teardownRecognition(); 1506 emitError(msg, type); 1507 setState('error'); 1508 const id = sessionId; 1509 settleTimer = setTimeout(() => { 1510 if (!destroyed && !legacy && id === sessionId && state === 'error') setState('idle'); 1511 }, ERROR_SETTLE_MS); 1512 } 1513 1514 // Fatal Web Speech failure: swap this controller to the WebSocket engine. 1515 // Silent from the consumer's point of view â no error state is emitted; a 1516 // running session resumes as connecting â listening on the new engine. 1517 function fallbackToLegacy() { 1518 if (destroyed || legacy) return; 1519 const wasRunning = active || state === 'connecting' || state === 'listening'; 1520 _webkitBroken = true; 1521 console.warn(LOG, 'Web Speech failed â falling back to WebSocket STT'); 1522 active = false; 1523 shouldRestart = false; 1524 teardownRecognition(); 1525 // Deliberate empty partial: the swap is silent (no error/idle event), so 1526 // nothing else clears a live interim preview. stt-ghost's showInterim('') 1527 // documents empty input as a full, unsealed teardown â the next partial 1528 // from the WebSocket engine rebuilds the preview. 1529 if (onPartial) { try { onPartial(''); } catch (e) {} } 1530 legacy = createWebSocketWithPreview({ ...config, lang }); 1531 if (wasRunning) legacy.start(); 1532 } 1533 1534 function buildRecognition(Ctor, id) { 1535 const rec = new Ctor(); 1536 1537 rec.continuous = true; 1538 rec.interimResults = true; 1539 rec.maxAlternatives = 1; 1540 rec.lang = RECOGNITION_LANGUAGE; 1541 1542 rec.onstart = () => { 1543 if (!isCurrent(rec, id) || !active) return; 1544 setState('listening'); 1545 }; 1546 1547 // First results index this instance has NOT yet delivered as final. 1548 // Chrome can point resultIndex back at already-finalized results; without 1549 // this high-water mark every such event re-emits the whole transcript 1550 // (the duplicated-words-while-speaking bug, harness-proven). Per instance: 1551 // a renewal starts a fresh recognition with a fresh results array. 1552 let finalizedThrough = 0; 1553 1554 rec.onresult = (event) => { 1555 // stop() asks the recognizer to finalize audio it already captured â 1556 // keep accepting those last results while the session winds down 1557 // (extension parity: dictation must not lose the user's final words). 1558 if (!isCurrent(rec, id) || (!active && !stoppingFlush)) return; 1559 1560 let interim = ''; 1561 for (let i = Math.max(event.resultIndex, finalizedThrough); i < event.results.length; i += 1) { 1562 const result = event.results[i]; 1563 const transcript = String(result?.[0]?.transcript || '').trim(); 1564 1565 if (result.isFinal) { 1566 finalizedThrough = i + 1; 1567 if (!transcript) continue; 1568 resetSilenceTimer(); 1569 if (onTranscript) onTranscript(normalizeSttText(transcript, NORMALIZE_LANG)); 1570 } else if (transcript) { 1571 interim += `${interim ? ' ' : ''}${transcript}`; 1572 } 1573 } 1574 1575 // Parity with the WebSocket rail: empty partials are never forwarded â 1576 // the ghost shell holds steady between utterances (finals clear their 1577 // own preview via the consumers' softClearInterim contract). 1578 if (interim) { 1579 resetSilenceTimer(); 1580 if (onPartial) onPartial(normalizeSttText(interim, NORMALIZE_LANG)); 1581 } 1582 }; 1583 1584 rec.onerror = (event) => { 1585 if (!isCurrent(rec, id)) return; 1586 const error = String(event?.error || 'unknown'); 1587 console.warn(LOG, 'recognition error:', error); 1588 1589 // Winding down after stop(): whatever the error, the session is over â 1590 // settle to idle (extension parity: a stopping session never surfaces 1591 // a late error state on a screen the user has moved on from). 1592 if (stoppingFlush) { 1593 finishStop(); 1594 return; 1595 } 1596 1597 // Chrome reports these right before onend during a quiet stretch; the 1598 // onend handler renews the instance and the session carries on. 1599 if (error === 'no-speech' || error === 'aborted') return; 1600 1601 if (error === 'not-allowed' || error === 'service-not-allowed') { 1602 failLocal('Giv mikrofontilladelse i din browser', 'permission'); 1603 return; 1604 } 1605 if (error === 'audio-capture') { 1606 failLocal('Ingen mikrofon fundet', 'no-device'); 1607 return; 1608 } 1609 1610 // network / language-not-supported / anything unexpected: the browser
1611 // engine is unusable here â hand the session to the WebSocket engine. 1612 fallbackToLegacy(); 1613 }; 1614 1615 rec.onend = () => { 1616 if (!isCurrent(rec, id)) return; 1617 1618 // The stop() flush is complete â the recognizer finalized what it had. 1619 if (stoppingFlush) { 1620 finishStop(); 1621 return; 1622 } 1623 1624 if (shouldRestart && active) { 1625 const lifetime = Date.now() - startedAt; 1626 rapidFailures = lifetime < RAPID_END_MS ? rapidFailures + 1 : 0; 1627 if (rapidFailures >= MAX_RAPID_FAILURES) { 1628 console.warn(LOG, 'recognition ended repeatedly'); 1629 fallbackToLegacy(); 1630 return; 1631 } 1632 1633 // Chrome closes continuous sessions periodically (~1 min). A fresh 1634 // instance keeps dictation running without any state flicker. 1635 recognition = null; 1636 launch(sessionId); 1637 return; 1638 } 1639 1640 recognition = null; 1641 if (state !== 'idle' && state !== 'error') setState('idle'); 1642 }; 1643 1644 return rec; 1645 } 1646 1647 function launch(id) { 1648 const Ctor = recognitionConstructor(); 1649 if (!Ctor) { 1650 fallbackToLegacy(); 1651 return; 1652 } 1653 1654 let rec; 1655 try { 1656 rec = buildRecognition(Ctor, id); 1657 } catch (err) { 1658 console.warn(LOG, 'could not create recognition:', err); 1659 fallbackToLegacy(); 1660 return; 1661 } 1662 1663 recognition = rec; 1664 startedAt = Date.now(); 1665 1666 try { 1667 rec.start(); 1668 } catch (err) { 1669 console.warn(LOG, 'could not start recognition:', err); 1670 if (err && err.name === 'NotAllowedError') { 1671 failLocal('Giv mikrofontilladelse i din browser', 'permission'); 1672 } else { 1673 fallbackToLegacy(); 1674 } 1675 } 1676 } 1677 1678 function startWebkit() { 1679 // A session still flushing its stop() cannot host a new start (extension 1680 // parity) â the next tap after the flush settles starts cleanly. 1681 if (destroyed || stoppingFlush || state === 'listening' || state === 'connecting') return; 1682 clearTimeout(settleTimer); 1683 sessionId++; 1684 active = true; 1685 shouldRestart = true; 1686 rapidFailures = 0; 1687 setState('connecting'); 1688 launch(sessionId); 1689 } 1690 1691 // End of a graceful stop: the recognizer has flushed (or the safety timer 1692 // fired). Invalidate the instance WITHOUT abort() so nothing is discarded. 1693 function finishStop() { 1694 clearTimeout(stopFlushTimer); 1695 stopFlushTimer = null; 1696 stoppingFlush = false; 1697 clearTimeout(silenceTimer); 1698 recognition = null; 1699 sessionId++; 1700 setState('idle'); 1701 } 1702 1703 function stopWebkit() { 1704 if (state === 'idle' || stoppingFlush) return; 1705 active = false; 1706 shouldRestart = false;
1707 clearTimeout(silenceTimer); 1708 1709 const rec = recognition; 1710 if (!rec) { 1711 setState('idle'); 1712 return; 1713 } 1714 1715 // Extension parity: stop() (never abort) lets Chrome return a final 1716 // result for audio captured before the user tapped stop. Those finals 1717 // flow through onresult during the flush; onend settles to idle. 1718 stoppingFlush = true; 1719 try { 1720 rec.stop(); 1721 } catch (err) { 1722 console.warn(LOG, 'stop failed:', err); 1723 stoppingFlush = false; 1724 teardownRecognition(); 1725 setState('idle'); 1726 return; 1727 } 1728 stopFlushTimer = setTimeout(() => { 1729 if (stoppingFlush) finishStop(); 1730 }, STOP_FLUSH_TIMEOUT_MS); 1731 } 1732 1733 return { 1734 start() { 1735 if (legacy) return legacy.start(); 1736 startWebkit(); 1737 }, 1738 stop() { 1739 if (legacy) return legacy.stop(); 1740 stopWebkit(); 1741 }, 1742 toggle() { 1743 if (legacy) return legacy.toggle(); 1744 if (state === 'listening' || state === 'connecting') stopWebkit(); 1745 else startWebkit(); 1746 }, 1747 destroy() { 1748 if (legacy) { legacy.destroy(); return; } 1749 destroyed = true; 1750 active = false; 1751 shouldRestart = false;
1752 teardownRecognition(); 1753 state = 'idle'; 1754 setButtonState('idle'); 1755 }, 1756 isActive() { 1757 if (legacy) return legacy.isActive(); 1758 return state === 'listening' || state === 'connecting'; 1759 }, 1760 setLang(l) { 1761 lang = l; 1762 if (legacy) legacy.setLang(l); 1763 }, 1764 }; 1765 1766 } 1767 1768 window.createStt = createStt; 1769 1770})();
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.