Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
94 changes: 65 additions & 29 deletions packages/voice-adapters/src/transcript-normalizer.ts
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,15 @@ interface Form {
term: RecognitionTerm;
rule: Rule;
}
/** A span of tokens `from..to` that matched one or more terms. */
interface Hit {
from: number;
to: number;
candidates: Form[];
near: boolean;
/** The shorter exact form at `from` that words running together into a longer term passed over. */
passedOver?: Hit | undefined;
}
interface Lexicon {
forms: Map<string, Form[]>;
longest: number;
Expand Down Expand Up @@ -115,24 +124,16 @@ export function normalizeTranscript(input: string, context: RecognitionContext):
const lexicon = lexiconFor(context);
const codeSwitched = VIETNAMESE_LETTERS.test(text);

type Hit = { from: number; to: number; candidates: Form[]; near: boolean };
const hits: Hit[] = [];
for (let index = 0; index < tokens.length; ) {
const hit = matchAt(tokens, text, index, lexicon);
if (hit === undefined) {
index += 1;
continue;
}
hits.push({ ...hit, candidates: withCaseVariants(hit.candidates, lexicon) });
index = hit.to;
}
const hits = hitsBetween(tokens, text, 0, tokens.length, lexicon);

// Evidence that the utterance is about code, by position, so a span never counts as its own evidence.
const anchors: number[] = [];
tokens.forEach((token, index) => {
if (TECHNICAL_CUES.has(token.lower)) anchors.push(index);
});
for (const hit of hits) {
// Words a run-together term absorbed, kept apart, since they support fewer spans: see `supported`.
const absorbed: number[] = [];
const addEvidence = (hit: Hit, into: number[]): void => {
const unique = distinctTerms(hit.candidates);
const source = text.slice(tokens[hit.from]!.start, tokens[hit.to - 1]!.end);
// Only a term heard in its own spelling, case aside, is evidence: a corrected span supporting another correction
Expand All @@ -145,10 +146,28 @@ export function normalizeTranscript(input: string, context: RecognitionContext):
source.toLowerCase() === term.text.toLowerCase() &&
unique.some((candidate) => source === candidate.text || isDistinctive(candidate))
) {
for (let at = hit.from; at < hit.to; at += 1) anchors.push(at);
for (let at = hit.from; at < hit.to; at += 1) into.push(at);
return;
}
}
const supported = (from: number, to: number): boolean => codeSwitched || anchors.some((at) => at < from || at >= to);
// Words that ran together into a longer term are evidence still, exactly as they would have been on their own:
// "web" heard exactly says the sentence is about code whether or not "web app" goes on to read as `webapp`. Only
// a shorter form starting the span opens it up; the exact matches after that form count too.
if (hit.passedOver !== undefined) {
addEvidence(hit.passedOver, absorbed);
for (const inner of hitsBetween(tokens, text, hit.passedOver.to, hit.to, lexicon)) addEvidence(inner, absorbed);
}
};
for (const hit of hits) addEvidence(hit, anchors);
const outside = (at: number, hit: Hit): boolean => at < hit.from || at >= hit.to;
// Absorbed words support only a span whose change no run-together span can rest on: one that absorbed no words and
// holds no evidence of its own. One that absorbed words too would take its own with it ("the web app and the web app"
// would vouch for itself). One holding evidence already supports every run-together span outside it, so their absorbed
// words are rewritten away: "Follow-up" heard for the tool follow-up, or the cue "code" in "code review" for
// `codeReview`, would turn "web app" into `webapp`, then be rewritten itself on the "web" that change took away.
const supported = (hit: Hit): boolean =>
codeSwitched ||
anchors.some((at) => outside(at, hit)) ||
(hit.passedOver === undefined && anchors.every((at) => outside(at, hit)) && absorbed.some((at) => outside(at, hit)));

const result: NormalizationResult = { text, changes: [], abstained: [], technical: [] };
const replacements: Array<{ start: number; end: number; to: string }> = [];
Expand Down Expand Up @@ -194,7 +213,7 @@ export function normalizeTranscript(input: string, context: RecognitionContext):
// word starting a sentence, so lowering it needs the same evidence as any plain word.
const needsContext =
rule !== "casing" || (lowers && isWordLikeTool(term)) || !(isDistinctive(term) || (lowers && term.kind === "command"));
if (needsContext && !supported(hit.from, hit.to)) continue;
if (needsContext && !supported(hit)) continue;
if (result.changes.length >= MAX_NORMALIZATION_CHANGES) continue;

replacements.push({ start, end, to: term.text });
Expand All @@ -213,33 +232,50 @@ export function normalizeTranscript(input: string, context: RecognitionContext):
return result;
}

/** Every match over tokens `from..end`, left to right, each with its spellings told apart only by case. */
function hitsBetween(tokens: readonly Token[], text: string, from: number, end: number, lexicon: Lexicon): Hit[] {
const hits: Hit[] = [];
for (let index = from; index < end; ) {
const hit = matchAt(tokens, text, index, end, lexicon);
if (hit === undefined) {
index += 1;
continue;
}
hits.push(withCaseVariantsOf(hit, lexicon));
index = hit.to;
}
return hits;
}

function withCaseVariantsOf(hit: Hit, lexicon: Lexicon): Hit {
const passedOver = hit.passedOver === undefined ? undefined : withCaseVariantsOf(hit.passedOver, lexicon);
return { ...hit, candidates: withCaseVariants(hit.candidates, lexicon), passedOver };
}

/**
* The longest exact match starting at `index`: a known form, or words that run together into a term. A near match is
* tried only when neither starts there.
* The longest exact match starting at `index` and ending by `end`: a known form, or words that run together into a
* term. A near match is tried only when neither starts there. When words run together into a term longer than the
* form starting there, the form is kept as `passedOver`, since what those words are on their own is still evidence.
*/
function matchAt(
tokens: readonly Token[],
text: string,
index: number,
lexicon: Lexicon,
): { from: number; to: number; candidates: Form[]; near: boolean } | undefined {
let form: { from: number; to: number; candidates: Form[]; near: boolean } | undefined;
for (let length = Math.min(lexicon.longest, tokens.length - index); length >= 1 && form === undefined; length -= 1) {
function matchAt(tokens: readonly Token[], text: string, index: number, end: number, lexicon: Lexicon): Hit | undefined {
let form: Hit | undefined;
for (let length = Math.min(lexicon.longest, end - index); length >= 1 && form === undefined; length -= 1) {
if (!joinedBySeparators(tokens, text, index, length)) continue;
const key = tokens.slice(index, index + length).map((token) => token.lower).join(" ");
const forms = lexicon.forms.get(key);
if (forms !== undefined) form = { from: index, to: index + length, candidates: forms, near: false };
}
for (let length = Math.min(3, tokens.length - index); length >= 1; length -= 1) {
for (let length = Math.min(3, end - index); length >= 1; length -= 1) {
// A form at least as long as the words left to try wins: the longer exact match is the one the person said.
if (form !== undefined && length <= form.to - index) return form;
if (!joinedBySeparators(tokens, text, index, length)) continue;
const compact = tokens.slice(index, index + length).map((token) => token.lower).join("");
// Words that run together into the term exactly ("clark cant web" for clarkcant-web) are a spacing variant, and
// one word longer than a form that starts the same way ("clark cant" for ClarkCant) is the longer term.
// longer than the form found at this position ("clark cant" for ClarkCant) they are the longer term.
const joined = length > 1 ? lexicon.compacts.filter((entry) => entry.compact === compact && entry.term.kind !== "command") : [];
if (joined.length > 0) {
return { from: index, to: index + length, candidates: joined.map((entry) => ({ term: entry.term, rule: "spacing" })), near: false };
const candidates = joined.map((entry): Form => ({ term: entry.term, rule: "spacing" }));
return { from: index, to: index + length, candidates, near: false, passedOver: form };
}
// A near match is a guess, and never outranks an exact form, however short.
if (form !== undefined) continue;
Expand Down
86 changes: 86 additions & 0 deletions packages/voice-adapters/test/transcript-normalizer.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -291,6 +291,92 @@ describe("words that run together into a longer term than a form starting the sa
expect(normalizeTranscript("sửa clark cant trước", session).text).toBe("sửa ClarkCant trước");
expect(normalizeTranscript("sửa clark cant web trước", session).text).toBe("sửa clarkcant-web trước");
});

describe("still counts the words a run-together term absorbed as evidence for the rest of the sentence", () => {
// `web` and `Web` are both real names, so "web" heard exactly is a known term and says the sentence is about code.
// Folded into `webapp`, it says so still: the run-together match is no weaker evidence than the words it took in.
const SESSION = buildRecognitionContext({
symbols: ["UserService", "userService", "Web", "web", "userserviceApi"],
packages: ["userservice-api", "webapp"],
});

it("restores ClarkCant beside a run-together web app, and leaves web app as heard", () => {
const result = normalizeTranscript("fix the clark cant web app now", SESSION);
expect(result.text).toBe("fix the ClarkCant web app now");
expect(result.changes).toEqual([{ from: "clark cant", to: "ClarkCant", rule: "spacing", kind: "repository" }]);
expect(result.abstained).toEqual([]);
});

it("restores React after a run-together web app", () => {
const result = normalizeTranscript("fix the web app react now", SESSION);
expect(result.text).toBe("fix the web app React now");
expect(result.changes).toEqual([{ from: "react", to: "React", rule: "casing", kind: "glossary" }]);
});

it("abstains on user service api, which reads as either run-together term", () => {
const result = normalizeTranscript("fix the user service api now", SESSION);
expect(result.text).toBe("fix the user service api now");
expect(result.changes).toEqual([]);
expect(result.abstained).toEqual([
{ start: 8, end: 24, text: "user service api", candidates: ["userservice-api", "userserviceApi"] },
]);
});

// A span rewritten into a run-together term would take its absorbed word with it, so that word never vouches for
// another span rewritten the same way: two guesses must not support each other.
it("leaves two run-together spans as heard when only each other's absorbed words support them", () => {
const session = buildRecognitionContext({ tools: ["web"], packages: ["webapp"] });
const result = normalizeTranscript("the web app and the web app", session);
expect(result.text).toBe("the web app and the web app");
expect(result.changes).toEqual([]);
});

it("leaves set up the web app for grandma as heard", () => {
const session = buildRecognitionContext({ tools: ["set", "web"], packages: ["setup", "webapp"] });
const result = normalizeTranscript("set up the web app for grandma", session);
expect(result.text).toBe("set up the web app for grandma");
expect(result.changes).toEqual([]);
});

it("leaves the front end of the web app as heard", () => {
const session = buildRecognitionContext({ tools: ["front", "web"], packages: ["frontend", "webapp"] });
const result = normalizeTranscript("the front end of the web app", session);
expect(result.text).toBe("the front end of the web app");
expect(result.changes).toEqual([]);
});

// A capitalised hyphenated or digit tool name is evidence, yet lowering it needs evidence too: the word a
// run-together span absorbed on its strength is not that evidence, or each change would rest only on the other.
it("keeps Follow-up when the only support for lowering it is the web that webapp absorbed", () => {
const session = buildRecognitionContext({ tools: ["follow-up", "web"], packages: ["webapp"] }, { glossary: false });
const result = normalizeTranscript("Follow-up on the web app", session);
expect(result.text).toBe("Follow-up on the webapp");
expect(result.changes).toEqual([{ from: "web app", to: "webapp", rule: "spacing", kind: "package" }]);
});

// A cue inside a span works the same way: "code" makes "web app" read as `webapp`, so the "web" it absorbed cannot
// in turn support "code review" becoming `codeReview`.
it("does not rewrite code review on the web that webapp absorbed after the cue in code review turned it", () => {
const session = buildRecognitionContext({ tools: ["web"], packages: ["webapp"], symbols: ["codeReview"] }, { glossary: false });
const result = normalizeTranscript("the code review of the web app", session);
expect(result.text).toBe("the code review of the webapp");
expect(result.changes).toEqual([{ from: "web app", to: "webapp", rule: "spacing", kind: "package" }]);
});

it("does not join type script on the web that webapp absorbed after the cue in type script turned it", () => {
const session = buildRecognitionContext({ tools: ["web"], packages: ["webapp"] });
const result = normalizeTranscript("type script web app", session);
expect(result.text).toBe("type script webapp");
expect(result.changes).toEqual([{ from: "web app", to: "webapp", rule: "spacing", kind: "package" }]);
});

it("keeps S3 when the only support for lowering it is the web that webapp absorbed", () => {
const session = buildRecognitionContext({ tools: ["s3", "web"], packages: ["webapp"] }, { glossary: false });
const result = normalizeTranscript("S3 is on the web app", session);
expect(result.text).toBe("S3 is on the webapp");
expect(result.changes).toEqual([{ from: "web app", to: "webapp", rule: "spacing", kind: "package" }]);
});
});
});

describe("a session term spelled one way beside the glossary's spelling", () => {
Expand Down
Loading