From 86907b0cce9a9a3f9932514d2fba389734ccb685 Mon Sep 17 00:00:00 2001 From: Duy Nguyen Date: Thu, 8 Oct 2026 04:56:52 +0700 Subject: [PATCH 1/4] fix(voice): count the words a run-together term absorbed as code evidence for the sentence When words ran together into a longer term, the shorter exact form they passed over no longer anchored the rest of an English sentence, so ClarkCant and React beside a run-together webapp were left as heard. The passed-over form and the exact matches inside the span now count as evidence, as they would on their own; they still never support the run-together span itself. Refs #608, #613 --- .../src/transcript-normalizer.ts | 76 ++++++++++++------- .../test/transcript-normalizer.spec.ts | 31 ++++++++ 2 files changed, 81 insertions(+), 26 deletions(-) diff --git a/packages/voice-adapters/src/transcript-normalizer.ts b/packages/voice-adapters/src/transcript-normalizer.ts index 0548ba5d..7942abb3 100644 --- a/packages/voice-adapters/src/transcript-normalizer.ts +++ b/packages/voice-adapters/src/transcript-normalizer.ts @@ -62,6 +62,15 @@ interface Form { term: RecognitionTerm; rule: Rule; } +/** A span of tokens `from..to` that matched one or more terms. */ +interface Hit { + from: number; + to: number; + candidates: Form[]; + near: boolean; + /** The shorter exact form at `from` that words running together into a longer term passed over. */ + passedOver?: Hit | undefined; +} interface Lexicon { forms: Map; longest: number; @@ -115,24 +124,14 @@ export function normalizeTranscript(input: string, context: RecognitionContext): const lexicon = lexiconFor(context); const codeSwitched = VIETNAMESE_LETTERS.test(text); - type Hit = { from: number; to: number; candidates: Form[]; near: boolean }; - const hits: Hit[] = []; - for (let index = 0; index < tokens.length; ) { - const hit = matchAt(tokens, text, index, lexicon); - if (hit === undefined) { - index += 1; - continue; - } - hits.push({ ...hit, candidates: withCaseVariants(hit.candidates, lexicon) }); - index = hit.to; - } + const hits = hitsBetween(tokens, text, 0, tokens.length, lexicon); // Evidence that the utterance is about code, by position, so a span never counts as its own evidence. const anchors: number[] = []; tokens.forEach((token, index) => { if (TECHNICAL_CUES.has(token.lower)) anchors.push(index); }); - for (const hit of hits) { + const addEvidence = (hit: Hit): void => { const unique = distinctTerms(hit.candidates); const source = text.slice(tokens[hit.from]!.start, tokens[hit.to - 1]!.end); // Only a term heard in its own spelling, case aside, is evidence: a corrected span supporting another correction @@ -146,8 +145,16 @@ export function normalizeTranscript(input: string, context: RecognitionContext): unique.some((candidate) => source === candidate.text || isDistinctive(candidate)) ) { for (let at = hit.from; at < hit.to; at += 1) anchors.push(at); + return; } - } + // Words that ran together into a longer term are evidence still, exactly as they would have been on their own: + // "web" heard exactly says the sentence is about code whether or not "web app" goes on to read as `webapp`. + if (hit.passedOver !== undefined) { + addEvidence(hit.passedOver); + for (const inner of hitsBetween(tokens, text, hit.passedOver.to, hit.to, lexicon)) addEvidence(inner); + } + }; + for (const hit of hits) addEvidence(hit); const supported = (from: number, to: number): boolean => codeSwitched || anchors.some((at) => at < from || at >= to); const result: NormalizationResult = { text, changes: [], abstained: [], technical: [] }; @@ -213,33 +220,50 @@ export function normalizeTranscript(input: string, context: RecognitionContext): return result; } +/** Every match over tokens `from..end`, left to right, each with its spellings told apart only by case. */ +function hitsBetween(tokens: readonly Token[], text: string, from: number, end: number, lexicon: Lexicon): Hit[] { + const hits: Hit[] = []; + for (let index = from; index < end; ) { + const hit = matchAt(tokens, text, index, end, lexicon); + if (hit === undefined) { + index += 1; + continue; + } + hits.push(withCaseVariantsOf(hit, lexicon)); + index = hit.to; + } + return hits; +} + +function withCaseVariantsOf(hit: Hit, lexicon: Lexicon): Hit { + const passedOver = hit.passedOver === undefined ? undefined : withCaseVariantsOf(hit.passedOver, lexicon); + return { ...hit, candidates: withCaseVariants(hit.candidates, lexicon), passedOver }; +} + /** - * The longest exact match starting at `index`: a known form, or words that run together into a term. A near match is - * tried only when neither starts there. + * The longest exact match starting at `index` and ending by `end`: a known form, or words that run together into a + * term. A near match is tried only when neither starts there. When words run together into a term longer than the + * form starting there, the form is kept as `passedOver`, since what those words are on their own is still evidence. */ -function matchAt( - tokens: readonly Token[], - text: string, - index: number, - lexicon: Lexicon, -): { from: number; to: number; candidates: Form[]; near: boolean } | undefined { - let form: { from: number; to: number; candidates: Form[]; near: boolean } | undefined; - for (let length = Math.min(lexicon.longest, tokens.length - index); length >= 1 && form === undefined; length -= 1) { +function matchAt(tokens: readonly Token[], text: string, index: number, end: number, lexicon: Lexicon): Hit | undefined { + let form: Hit | undefined; + for (let length = Math.min(lexicon.longest, end - index); length >= 1 && form === undefined; length -= 1) { if (!joinedBySeparators(tokens, text, index, length)) continue; const key = tokens.slice(index, index + length).map((token) => token.lower).join(" "); const forms = lexicon.forms.get(key); if (forms !== undefined) form = { from: index, to: index + length, candidates: forms, near: false }; } - for (let length = Math.min(3, tokens.length - index); length >= 1; length -= 1) { + for (let length = Math.min(3, end - index); length >= 1; length -= 1) { // A form at least as long as the words left to try wins: the longer exact match is the one the person said. if (form !== undefined && length <= form.to - index) return form; if (!joinedBySeparators(tokens, text, index, length)) continue; const compact = tokens.slice(index, index + length).map((token) => token.lower).join(""); // Words that run together into the term exactly ("clark cant web" for clarkcant-web) are a spacing variant, and - // one word longer than a form that starts the same way ("clark cant" for ClarkCant) is the longer term. + // longer than the form found at this position ("clark cant" for ClarkCant) they are the longer term. const joined = length > 1 ? lexicon.compacts.filter((entry) => entry.compact === compact && entry.term.kind !== "command") : []; if (joined.length > 0) { - return { from: index, to: index + length, candidates: joined.map((entry) => ({ term: entry.term, rule: "spacing" })), near: false }; + const candidates = joined.map((entry): Form => ({ term: entry.term, rule: "spacing" })); + return { from: index, to: index + length, candidates, near: false, passedOver: form }; } // A near match is a guess, and never outranks an exact form, however short. if (form !== undefined) continue; diff --git a/packages/voice-adapters/test/transcript-normalizer.spec.ts b/packages/voice-adapters/test/transcript-normalizer.spec.ts index 95cca8d1..25cfe49e 100644 --- a/packages/voice-adapters/test/transcript-normalizer.spec.ts +++ b/packages/voice-adapters/test/transcript-normalizer.spec.ts @@ -291,6 +291,37 @@ describe("words that run together into a longer term than a form starting the sa expect(normalizeTranscript("sửa clark cant trước", session).text).toBe("sửa ClarkCant trước"); expect(normalizeTranscript("sửa clark cant web trước", session).text).toBe("sửa clarkcant-web trước"); }); + + describe("still counts the words a run-together term absorbed as evidence for the rest of the sentence", () => { + // `web` and `Web` are both real names, so "web" heard exactly is a known term and says the sentence is about code. + // Folded into `webapp`, it says so still: the run-together match is no weaker evidence than the words it took in. + const SESSION = buildRecognitionContext({ + symbols: ["UserService", "userService", "Web", "web", "userserviceApi"], + packages: ["userservice-api", "webapp"], + }); + + it("restores ClarkCant beside a run-together web app, and leaves web app as heard", () => { + const result = normalizeTranscript("fix the clark cant web app now", SESSION); + expect(result.text).toBe("fix the ClarkCant web app now"); + expect(result.changes).toEqual([{ from: "clark cant", to: "ClarkCant", rule: "spacing", kind: "repository" }]); + expect(result.abstained).toEqual([]); + }); + + it("restores React after a run-together web app", () => { + const result = normalizeTranscript("fix the web app react now", SESSION); + expect(result.text).toBe("fix the web app React now"); + expect(result.changes).toEqual([{ from: "react", to: "React", rule: "casing", kind: "glossary" }]); + }); + + it("abstains on user service api, which reads as either run-together term", () => { + const result = normalizeTranscript("fix the user service api now", SESSION); + expect(result.text).toBe("fix the user service api now"); + expect(result.changes).toEqual([]); + expect(result.abstained).toEqual([ + { start: 8, end: 24, text: "user service api", candidates: ["userservice-api", "userserviceApi"] }, + ]); + }); + }); }); describe("a session term spelled one way beside the glossary's spelling", () => { From 66212bc5adfad6a841e6ef8d4a1df30681fc67a7 Mon Sep 17 00:00:00 2001 From: Duy Nguyen Date: Thu, 8 Oct 2026 05:10:04 +0700 Subject: [PATCH 2/4] fix(voice): keep two run-together spans from vouching for each other A word a run-together term absorbed now supports only spans that did not absorb a word themselves. Such a span is rewritten only on a technical cue or a term heard in its own spelling, so "the web app and the web app" and "set up the web app for grandma" stay as heard, while "fix the clark cant web app now" still gives ClarkCant and "fix the web app react now" still restores React. --- .../src/transcript-normalizer.ts | 25 +++++++++++++------ .../test/transcript-normalizer.spec.ts | 23 +++++++++++++++++ 2 files changed, 40 insertions(+), 8 deletions(-) diff --git a/packages/voice-adapters/src/transcript-normalizer.ts b/packages/voice-adapters/src/transcript-normalizer.ts index 7942abb3..fca1ae90 100644 --- a/packages/voice-adapters/src/transcript-normalizer.ts +++ b/packages/voice-adapters/src/transcript-normalizer.ts @@ -131,7 +131,9 @@ export function normalizeTranscript(input: string, context: RecognitionContext): tokens.forEach((token, index) => { if (TECHNICAL_CUES.has(token.lower)) anchors.push(index); }); - const addEvidence = (hit: Hit): void => { + // Words a run-together term absorbed, kept apart: they support every other span except one that absorbed words too. + const absorbed: number[] = []; + const addEvidence = (hit: Hit, into: number[]): void => { const unique = distinctTerms(hit.candidates); const source = text.slice(tokens[hit.from]!.start, tokens[hit.to - 1]!.end); // Only a term heard in its own spelling, case aside, is evidence: a corrected span supporting another correction @@ -144,18 +146,25 @@ export function normalizeTranscript(input: string, context: RecognitionContext): source.toLowerCase() === term.text.toLowerCase() && unique.some((candidate) => source === candidate.text || isDistinctive(candidate)) ) { - for (let at = hit.from; at < hit.to; at += 1) anchors.push(at); + for (let at = hit.from; at < hit.to; at += 1) into.push(at); return; } // Words that ran together into a longer term are evidence still, exactly as they would have been on their own: - // "web" heard exactly says the sentence is about code whether or not "web app" goes on to read as `webapp`. + // "web" heard exactly says the sentence is about code whether or not "web app" goes on to read as `webapp`. Only + // a shorter form starting the span opens it up; the exact matches after that form count too. if (hit.passedOver !== undefined) { - addEvidence(hit.passedOver); - for (const inner of hitsBetween(tokens, text, hit.passedOver.to, hit.to, lexicon)) addEvidence(inner); + addEvidence(hit.passedOver, absorbed); + for (const inner of hitsBetween(tokens, text, hit.passedOver.to, hit.to, lexicon)) addEvidence(inner, absorbed); } }; - for (const hit of hits) addEvidence(hit); - const supported = (from: number, to: number): boolean => codeSwitched || anchors.some((at) => at < from || at >= to); + for (const hit of hits) addEvidence(hit, anchors); + const outside = (at: number, hit: Hit): boolean => at < hit.from || at >= hit.to; + // A span that absorbed words is rewritten only on evidence that is not itself a guess: another such span's absorbed + // word would be rewritten away too, and the two ("the web app and the web app") would vouch for each other. + const supported = (hit: Hit): boolean => + codeSwitched || + anchors.some((at) => outside(at, hit)) || + (hit.passedOver === undefined && absorbed.some((at) => outside(at, hit))); const result: NormalizationResult = { text, changes: [], abstained: [], technical: [] }; const replacements: Array<{ start: number; end: number; to: string }> = []; @@ -201,7 +210,7 @@ export function normalizeTranscript(input: string, context: RecognitionContext): // word starting a sentence, so lowering it needs the same evidence as any plain word. const needsContext = rule !== "casing" || (lowers && isWordLikeTool(term)) || !(isDistinctive(term) || (lowers && term.kind === "command")); - if (needsContext && !supported(hit.from, hit.to)) continue; + if (needsContext && !supported(hit)) continue; if (result.changes.length >= MAX_NORMALIZATION_CHANGES) continue; replacements.push({ start, end, to: term.text }); diff --git a/packages/voice-adapters/test/transcript-normalizer.spec.ts b/packages/voice-adapters/test/transcript-normalizer.spec.ts index 25cfe49e..8f35e147 100644 --- a/packages/voice-adapters/test/transcript-normalizer.spec.ts +++ b/packages/voice-adapters/test/transcript-normalizer.spec.ts @@ -321,6 +321,29 @@ describe("words that run together into a longer term than a form starting the sa { start: 8, end: 24, text: "user service api", candidates: ["userservice-api", "userserviceApi"] }, ]); }); + + // A span rewritten into a run-together term would take its absorbed word with it, so that word never vouches for + // another span rewritten the same way: two guesses must not support each other. + it("leaves two run-together spans as heard when only each other's absorbed words support them", () => { + const session = buildRecognitionContext({ tools: ["web"], packages: ["webapp"] }); + const result = normalizeTranscript("the web app and the web app", session); + expect(result.text).toBe("the web app and the web app"); + expect(result.changes).toEqual([]); + }); + + it("leaves set up the web app for grandma as heard", () => { + const session = buildRecognitionContext({ tools: ["set", "web"], packages: ["setup", "webapp"] }); + const result = normalizeTranscript("set up the web app for grandma", session); + expect(result.text).toBe("set up the web app for grandma"); + expect(result.changes).toEqual([]); + }); + + it("leaves the front end of the web app as heard", () => { + const session = buildRecognitionContext({ tools: ["front", "web"], packages: ["frontend", "webapp"] }); + const result = normalizeTranscript("the front end of the web app", session); + expect(result.text).toBe("the front end of the web app"); + expect(result.changes).toEqual([]); + }); }); }); From dd667399c76682fb8317115eee7706ddb0dfd7e5 Mon Sep 17 00:00:00 2001 From: Duy Nguyen Date: Thu, 8 Oct 2026 05:17:54 +0700 Subject: [PATCH 3/4] fix(voice): keep a rewritten evidence term from leaning on a word it let a span absorb A term that counts as evidence while it is itself rewritten, such as "Follow-up" heard for the tool follow-up or "S3" for s3, supports a run-together span that absorbs a word like "web". That absorbed word no longer supports lowering the term in return, so "Follow-up on the web app" keeps "Follow-up" as main does, instead of each change resting only on the other. --- .../voice-adapters/src/transcript-normalizer.ts | 17 ++++++++++------- .../test/transcript-normalizer.spec.ts | 16 ++++++++++++++++ 2 files changed, 26 insertions(+), 7 deletions(-) diff --git a/packages/voice-adapters/src/transcript-normalizer.ts b/packages/voice-adapters/src/transcript-normalizer.ts index fca1ae90..d3db66a9 100644 --- a/packages/voice-adapters/src/transcript-normalizer.ts +++ b/packages/voice-adapters/src/transcript-normalizer.ts @@ -131,9 +131,9 @@ export function normalizeTranscript(input: string, context: RecognitionContext): tokens.forEach((token, index) => { if (TECHNICAL_CUES.has(token.lower)) anchors.push(index); }); - // Words a run-together term absorbed, kept apart: they support every other span except one that absorbed words too. + // Words a run-together term absorbed, kept apart, since they support fewer spans: see `supported`. const absorbed: number[] = []; - const addEvidence = (hit: Hit, into: number[]): void => { + const addEvidence = (hit: Hit, into: number[]): boolean => { const unique = distinctTerms(hit.candidates); const source = text.slice(tokens[hit.from]!.start, tokens[hit.to - 1]!.end); // Only a term heard in its own spelling, case aside, is evidence: a corrected span supporting another correction @@ -147,7 +147,7 @@ export function normalizeTranscript(input: string, context: RecognitionContext): unique.some((candidate) => source === candidate.text || isDistinctive(candidate)) ) { for (let at = hit.from; at < hit.to; at += 1) into.push(at); - return; + return true; } // Words that ran together into a longer term are evidence still, exactly as they would have been on their own: // "web" heard exactly says the sentence is about code whether or not "web app" goes on to read as `webapp`. Only @@ -156,15 +156,18 @@ export function normalizeTranscript(input: string, context: RecognitionContext): addEvidence(hit.passedOver, absorbed); for (const inner of hitsBetween(tokens, text, hit.passedOver.to, hit.to, lexicon)) addEvidence(inner, absorbed); } + return false; }; - for (const hit of hits) addEvidence(hit, anchors); + const evidence = new Set(hits.filter((hit) => addEvidence(hit, anchors))); const outside = (at: number, hit: Hit): boolean => at < hit.from || at >= hit.to; - // A span that absorbed words is rewritten only on evidence that is not itself a guess: another such span's absorbed - // word would be rewritten away too, and the two ("the web app and the web app") would vouch for each other. + // Absorbed words support only a span whose change no run-together span can rest on. One that absorbed words too would + // take its own with it ("the web app and the web app" would vouch for itself), and so would a term that is evidence + // while rewritten: "Follow-up" heard for the tool follow-up would turn "web app" into `webapp`, then lower itself on + // the "web" that rewrite took away. const supported = (hit: Hit): boolean => codeSwitched || anchors.some((at) => outside(at, hit)) || - (hit.passedOver === undefined && absorbed.some((at) => outside(at, hit))); + (hit.passedOver === undefined && !evidence.has(hit) && absorbed.some((at) => outside(at, hit))); const result: NormalizationResult = { text, changes: [], abstained: [], technical: [] }; const replacements: Array<{ start: number; end: number; to: string }> = []; diff --git a/packages/voice-adapters/test/transcript-normalizer.spec.ts b/packages/voice-adapters/test/transcript-normalizer.spec.ts index 8f35e147..6e30dd34 100644 --- a/packages/voice-adapters/test/transcript-normalizer.spec.ts +++ b/packages/voice-adapters/test/transcript-normalizer.spec.ts @@ -344,6 +344,22 @@ describe("words that run together into a longer term than a form starting the sa expect(result.text).toBe("the front end of the web app"); expect(result.changes).toEqual([]); }); + + // A capitalised hyphenated or digit tool name is evidence, yet lowering it needs evidence too: the word a + // run-together span absorbed on its strength is not that evidence, or each change would rest only on the other. + it("keeps Follow-up when the only support for lowering it is the web that webapp absorbed", () => { + const session = buildRecognitionContext({ tools: ["follow-up", "web"], packages: ["webapp"] }, { glossary: false }); + const result = normalizeTranscript("Follow-up on the web app", session); + expect(result.text).toBe("Follow-up on the webapp"); + expect(result.changes).toEqual([{ from: "web app", to: "webapp", rule: "spacing", kind: "package" }]); + }); + + it("keeps S3 when the only support for lowering it is the web that webapp absorbed", () => { + const session = buildRecognitionContext({ tools: ["s3", "web"], packages: ["webapp"] }, { glossary: false }); + const result = normalizeTranscript("S3 is on the web app", session); + expect(result.text).toBe("S3 is on the webapp"); + expect(result.changes).toEqual([{ from: "web app", to: "webapp", rule: "spacing", kind: "package" }]); + }); }); }); From a08285bf600109ede2ea9eba72e6dd032ec41a15 Mon Sep 17 00:00:00 2001 From: Duy Nguyen Date: Thu, 8 Oct 2026 05:27:16 +0700 Subject: [PATCH 4/4] fix(voice): keep absorbed words from supporting a span that holds evidence itself A span that holds a coding cue or an evidence term already supports every run-together span outside it, so the words those spans absorbed are rewritten away. They no longer support that span in return: "the code review of the web app" keeps "code review" and "type script web app" keeps "type script", as main does. --- .../src/transcript-normalizer.ts | 18 +++++++++--------- .../test/transcript-normalizer.spec.ts | 16 ++++++++++++++++ 2 files changed, 25 insertions(+), 9 deletions(-) diff --git a/packages/voice-adapters/src/transcript-normalizer.ts b/packages/voice-adapters/src/transcript-normalizer.ts index d3db66a9..a2b49cb8 100644 --- a/packages/voice-adapters/src/transcript-normalizer.ts +++ b/packages/voice-adapters/src/transcript-normalizer.ts @@ -133,7 +133,7 @@ export function normalizeTranscript(input: string, context: RecognitionContext): }); // Words a run-together term absorbed, kept apart, since they support fewer spans: see `supported`. const absorbed: number[] = []; - const addEvidence = (hit: Hit, into: number[]): boolean => { + const addEvidence = (hit: Hit, into: number[]): void => { const unique = distinctTerms(hit.candidates); const source = text.slice(tokens[hit.from]!.start, tokens[hit.to - 1]!.end); // Only a term heard in its own spelling, case aside, is evidence: a corrected span supporting another correction @@ -147,7 +147,7 @@ export function normalizeTranscript(input: string, context: RecognitionContext): unique.some((candidate) => source === candidate.text || isDistinctive(candidate)) ) { for (let at = hit.from; at < hit.to; at += 1) into.push(at); - return true; + return; } // Words that ran together into a longer term are evidence still, exactly as they would have been on their own: // "web" heard exactly says the sentence is about code whether or not "web app" goes on to read as `webapp`. Only @@ -156,18 +156,18 @@ export function normalizeTranscript(input: string, context: RecognitionContext): addEvidence(hit.passedOver, absorbed); for (const inner of hitsBetween(tokens, text, hit.passedOver.to, hit.to, lexicon)) addEvidence(inner, absorbed); } - return false; }; - const evidence = new Set(hits.filter((hit) => addEvidence(hit, anchors))); + for (const hit of hits) addEvidence(hit, anchors); const outside = (at: number, hit: Hit): boolean => at < hit.from || at >= hit.to; - // Absorbed words support only a span whose change no run-together span can rest on. One that absorbed words too would - // take its own with it ("the web app and the web app" would vouch for itself), and so would a term that is evidence - // while rewritten: "Follow-up" heard for the tool follow-up would turn "web app" into `webapp`, then lower itself on - // the "web" that rewrite took away. + // Absorbed words support only a span whose change no run-together span can rest on: one that absorbed no words and + // holds no evidence of its own. One that absorbed words too would take its own with it ("the web app and the web app" + // would vouch for itself). One holding evidence already supports every run-together span outside it, so their absorbed + // words are rewritten away: "Follow-up" heard for the tool follow-up, or the cue "code" in "code review" for + // `codeReview`, would turn "web app" into `webapp`, then be rewritten itself on the "web" that change took away. const supported = (hit: Hit): boolean => codeSwitched || anchors.some((at) => outside(at, hit)) || - (hit.passedOver === undefined && !evidence.has(hit) && absorbed.some((at) => outside(at, hit))); + (hit.passedOver === undefined && anchors.every((at) => outside(at, hit)) && absorbed.some((at) => outside(at, hit))); const result: NormalizationResult = { text, changes: [], abstained: [], technical: [] }; const replacements: Array<{ start: number; end: number; to: string }> = []; diff --git a/packages/voice-adapters/test/transcript-normalizer.spec.ts b/packages/voice-adapters/test/transcript-normalizer.spec.ts index 6e30dd34..fe4eea2c 100644 --- a/packages/voice-adapters/test/transcript-normalizer.spec.ts +++ b/packages/voice-adapters/test/transcript-normalizer.spec.ts @@ -354,6 +354,22 @@ describe("words that run together into a longer term than a form starting the sa expect(result.changes).toEqual([{ from: "web app", to: "webapp", rule: "spacing", kind: "package" }]); }); + // A cue inside a span works the same way: "code" makes "web app" read as `webapp`, so the "web" it absorbed cannot + // in turn support "code review" becoming `codeReview`. + it("does not rewrite code review on the web that webapp absorbed after the cue in code review turned it", () => { + const session = buildRecognitionContext({ tools: ["web"], packages: ["webapp"], symbols: ["codeReview"] }, { glossary: false }); + const result = normalizeTranscript("the code review of the web app", session); + expect(result.text).toBe("the code review of the webapp"); + expect(result.changes).toEqual([{ from: "web app", to: "webapp", rule: "spacing", kind: "package" }]); + }); + + it("does not join type script on the web that webapp absorbed after the cue in type script turned it", () => { + const session = buildRecognitionContext({ tools: ["web"], packages: ["webapp"] }); + const result = normalizeTranscript("type script web app", session); + expect(result.text).toBe("type script webapp"); + expect(result.changes).toEqual([{ from: "web app", to: "webapp", rule: "spacing", kind: "package" }]); + }); + it("keeps S3 when the only support for lowering it is the web that webapp absorbed", () => { const session = buildRecognitionContext({ tools: ["s3", "web"], packages: ["webapp"] }, { glossary: false }); const result = normalizeTranscript("S3 is on the web app", session);