From 494fbe20b758b09514e1b60ab58e2ece1136ce8b Mon Sep 17 00:00:00 2001 From: Joshua Horton Date: Tue, 23 Sep 2025 14:54:59 -0500 Subject: [PATCH] feat(web): define keying function that matches paths that result in compatible context tokenizations This method is designed to identify changes in tokenization upon input that result in the same tokenization effects when applied. Such cases are safe to batch together when building search path extensions. Build-bot: skip build:web Test-bot: skip --- .../main/correction/tokenization-subsets.ts | 83 +++ .../worker-thread/src/main/test-index.ts | 1 + .../context/tokenization-subsets.tests.ts | 513 ++++++++++++++++++ 3 files changed, 597 insertions(+) create mode 100644 web/src/engine/predictive-text/worker-thread/src/main/correction/tokenization-subsets.ts create mode 100644 web/src/test/auto/headless/engine/predictive-text/worker-thread/context/tokenization-subsets.tests.ts diff --git a/web/src/engine/predictive-text/worker-thread/src/main/correction/tokenization-subsets.ts b/web/src/engine/predictive-text/worker-thread/src/main/correction/tokenization-subsets.ts new file mode 100644 index 0000000000..2e9304ece9 --- /dev/null +++ b/web/src/engine/predictive-text/worker-thread/src/main/correction/tokenization-subsets.ts @@ -0,0 +1,83 @@ +import { SENTINEL_CODE_UNIT } from '@keymanapp/models-templates'; +import { KMWString } from '@keymanapp/web-utils'; + +import { TokenizationTransitionEdits } from './context-tokenization.js'; + +export function precomputationSubsetKeyer(tokenizationEdits: TokenizationTransitionEdits): string { + const { alignment, tokenizedTransform } = tokenizationEdits; + const { edgeWindow, merges, splits, unmappedEdits } = alignment; + const components: string[] = []; + + // First entry: based on the edge window. The real key: what's the edit + // boundary? We need to apply to the same token and portion thereof. + const editBoundary = edgeWindow.editBoundary; + + // It's not about the boundary text - we just need to ensure it's the 'same' + // token - comprised of the same keystrokes. `sourceRangeKey` reflects the + // actual input for the source keystrokes. We might have deleted part of it + // in this tokenization, but that doesn't matter here - we want to imply the + // represented keystroke range. + const boundaryEdgeIndex = editBoundary.tokenIndex - edgeWindow.sliceIndex; + const boundaryComponent = `B${editBoundary.tokenIndex}=${editBoundary.sourceRangeKey}`; + + components.push(boundaryComponent); + + // Identify the new boundary token's length - as it appears after any related + // merges or splits. + let boundaryTextLen = KMWString.length(editBoundary.text); + const boundaryMerge = merges.find((m) => m.inputs.find(i => i.index == boundaryEdgeIndex)); + const boundarySplit = splits.find((s) => s.input.index == boundaryEdgeIndex); + if(boundaryMerge) { + boundaryTextLen = KMWString.length(boundaryMerge.match.text); + } else if(boundarySplit) { + boundaryTextLen = KMWString.length(boundarySplit.matches[boundarySplit.matches.length - 1].text); + } + + // Now, based on the transform tokenization. We want to force uniqueness for + // all variations of result length on each tokenized transform resulting from + // the precomputation's represented keystroke. + for(const {0: relativeIndex, 1: transform} of tokenizedTransform.entries()) { + const insertLen = KMWString.length(transform.insert); + if(relativeIndex > 0) { + // The true boundary lie before the insert if the value is non-zero; + // don't differentiate here! + boundaryTextLen = 0; + } + + if(boundaryTextLen) { + // transform.deleteLeft was already handled during boundary computation - + // do not include it here! + components.push(`BI@${relativeIndex}-${boundaryTextLen + insertLen}`); + boundaryTextLen = 0; + } else { + components.push(`I@${relativeIndex}-${insertLen}`); + } + } + + if(merges.length > 0) { + components.push('M:' + merges.map((matchMap) => { + // Text may be more unique, but is likely unnecessary; index yields shorter, + // easier to process keys. + const inputPortion = matchMap.inputs.map(i => '' + i.index).join('+'); + return `M:${inputPortion}=>${matchMap.match.index}`; + }).join(',')); + } + + if(splits.length > 0) { + components.push('S:' + splits.map((matchMap) => { + // Text may be more unique, but is likely unnecessary; index yields shorter, + // easier to process keys. + const matchPortion = matchMap.matches.map(m => '' + m.index).join('+'); + return `${matchMap.input.index}=>${matchPortion}`; + }).join(',')); + } + + if(unmappedEdits.length > 0) { + // We really shouldn't have these, let alone often. + components.push('UE:' + unmappedEdits.map((edit) => { + return `${edit.op}(${edit.input ?? ''}-${edit.match ?? ''}`; + }).join(',')); + } + + return components.join(SENTINEL_CODE_UNIT); +} diff --git a/web/src/engine/predictive-text/worker-thread/src/main/test-index.ts b/web/src/engine/predictive-text/worker-thread/src/main/test-index.ts index b17e9c598d..2c64db440b 100644 --- a/web/src/engine/predictive-text/worker-thread/src/main/test-index.ts +++ b/web/src/engine/predictive-text/worker-thread/src/main/test-index.ts @@ -6,6 +6,7 @@ export { ContextTracker } from './correction/context-tracker.js'; export { ContextTransition } from './correction/context-transition.js'; export * from './correction/alignment-helpers.js'; export { ExtendedEditOperation, SegmentableDistanceCalculation } from './correction/segmentable-calculation.js'; +export * from './correction/tokenization-subsets.js'; export * as correction from './correction/index.js'; export * from './model-helpers.js'; export * as models from './models/index.js'; diff --git a/web/src/test/auto/headless/engine/predictive-text/worker-thread/context/tokenization-subsets.tests.ts b/web/src/test/auto/headless/engine/predictive-text/worker-thread/context/tokenization-subsets.tests.ts new file mode 100644 index 0000000000..132117812e --- /dev/null +++ b/web/src/test/auto/headless/engine/predictive-text/worker-thread/context/tokenization-subsets.tests.ts @@ -0,0 +1,513 @@ +/* + * Keyman is copyright (C) SIL Global. MIT License. + * + * Created by jahorton on 2025-09-23 + * + * This file contains low-level tests designed to validate the behavior of the + * of the ContextTokenization class and its integration with the lower-level + * classes that it utilizes. + */ + +import { assert } from 'chai'; + +import { default as defaultBreaker } from '@keymanapp/models-wordbreakers'; +import { LexicalModelTypes } from '@keymanapp/common-types'; +import { deepCopy } from '@keymanapp/web-utils'; +import { jsonFixture } from '@keymanapp/common-test-resources/model-helpers.mjs'; + +import { buildEdgeWindow, ContextToken, ContextTokenization, models, precomputationSubsetKeyer, TokenizationTransitionEdits } from '@keymanapp/lm-worker/test-index'; + +import Transform = LexicalModelTypes.Transform; +import TrieModel = models.TrieModel; + +var plainModel = new TrieModel(jsonFixture('models/tries/english-1000'), + {wordBreaker: defaultBreaker}); + +function toToken(text: string) { + let isWhitespace = text == ' '; + let token = new ContextToken(plainModel, text); + token.isWhitespace = isWhitespace; + return token; +} + +describe('precomputationSubsetKeyer', function() { + it("safely generates keys for empty transition + empty contexts", () => { + const rawTextTokens = ['']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: '', deleteLeft: 0 }); + return map; + })() + }; + const key = precomputationSubsetKeyer(precomputation1); + + assert.isOk(key); + }); + + it("generates different keys for transforms of different insert lengths on the same context", () => { + const rawTextTokens = ['an', ' ', 'apple', ' ', 'a', ' ', 'day']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: '', deleteLeft: 0 }); + return map; + })() + }; + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 's', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: 's', deleteLeft: 0 }); + return map; + })() + + assert.deepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + assert.notEqual(key2, key1); + }); + + it("generates different keys for transforms of different deleteLeft lengths on the same context", () => { + const rawTextTokens = ['an', ' ', 'apple', ' ', 'a', ' ', 'day']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 's', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: 's', deleteLeft: 0 }); + return map; + })() + }; + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 'b', deleteLeft: 1, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: 'b', deleteLeft: 1 }); + return map; + })() + + assert.notDeepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + assert.notEqual(key2, key1); + }); + + it("generates matching keys when boundary token + transform results in equal length token with same source text (1)", () => { + const rawTextTokens = ['an', ' ', 'apple', ' ', 'a', ' ']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + // concept: inputs were 'd', 'a', 'te', 's' + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + [...tokenization.tokens, (() => { + const token = new ContextToken(plainModel, 'da'); + // source text: 'date' + token.addInput( + {trueTransform: {insert: 'te', deleteLeft: 0}, inputStartIndex: 0}, + [{sample: {insert: 'te', deleteLeft: 0}, p: 1}] + ); + return token; + })()], + { insert: 's', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: 's', deleteLeft: 0 }); + return map; + })() + }; + + // concept: inputs were 'd', 'a', 't', 'es' + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + [...tokenization.tokens, (() => { + const token = new ContextToken(plainModel, 'da'); + // source text: 'date' + token.addInput( + {trueTransform: {insert: 'te', deleteLeft: 0}, inputStartIndex: 0}, + [{sample: {insert: 't', deleteLeft: 0}, p: 1}] + ); + return token; + })()], + { insert: 'es', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: 'es', deleteLeft: 0 }); + return map; + })() + + assert.notDeepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + assert.equal(key2, key1); + }); + + // - also due to deleteLeft effects: 2 + 1 vs a 2 + 2-dl:1 + it("generates matching keys when boundary token + transform results in equal length token with same source text (1)", () => { + const rawTextTokens = ['an', ' ', 'apple', ' ', 'a', ' ']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + // concept: inputs were 'd', 'a', 'ts', 'e' (with delete-left 1) + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + [...tokenization.tokens, (() => { + const token = new ContextToken(plainModel, 'da'); + token.isPartial = true; + // source text: 'dat' + token.addInput( + {trueTransform: {insert: 't', deleteLeft: 0}, inputStartIndex: 0}, + [{sample: {insert: 'ts', deleteLeft: 0}, p: 1}] + ); + return token; + })()], + { insert: 'e', deleteLeft: 1, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: 'e', deleteLeft: 1 }); + return map; + })() + }; + + // concept: inputs were 'd', 'a', 't', 'e' + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + [...tokenization.tokens, (() => { + const token = new ContextToken(plainModel, 'da'); + token.isPartial = true; + // source text: 'dat' + token.addInput( + {trueTransform: {insert: 't', deleteLeft: 0}, inputStartIndex: 0}, + [{sample: {insert: 't', deleteLeft: 0}, p: 1}] + ); + return token; + })()], + { insert: 'e', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: 'e', deleteLeft: 0 }); + return map; + })() + + // delete lengths differ, but that should be it. + assert.notDeepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const alteredEdgeWindow = { + ...precomputation2.alignment.edgeWindow, + deleteLengths: [1] + } + assert.deepEqual(alteredEdgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + // Both result in the same net length of token (4) starting at the same + // point in the context / at the same keystroke. + assert.equal(key2, key1); + }); + + it("properly notes new boundary token length on boundary-final merge", () => { + const rawTextTokens = ['she', ' ', 'says', ' ', 'I', ' ', 'can', '\'']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const edgeWindow1 = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 't', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [{ + // The indices specified here are the edge-window-internal indices. + inputs: [ + { text: 'can', index: rawTextTokens.length - 2 - edgeWindow1.sliceIndex }, + { text: '\'', index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex } + ], match: { + text: 'can\'', + index: rawTextTokens.length - 2 - edgeWindow1.sliceIndex + } + }], + splits: [], + unmappedEdits: [], + edgeWindow: edgeWindow1 + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: 't', deleteLeft: 0 }); + return map; + })() + }; + + const key1 = precomputationSubsetKeyer(precomputation1); + // Boundary token says is length 2 - as if appending to just `'`, not to `can'`. + assert.isFalse(key1.indexOf("BI@0-2") > -1, "The key's merge marker length does not correspond to merged token length"); + // Boundary token says is length 5 - as if appending to `can'`. + assert.isTrue(key1.indexOf("BI@0-5") > -1); + }); + + it("generates different keys for matching transforms when one causes token merge", () => { + const rawTextTokens = ['she', ' ', 'says', ' ', 'I', ' ', 'can', '\'']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const edgeWindow1 = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 't', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [{ + // The indices specified here are the edge-window-internal indices. + inputs: [ + { text: 'can', index: rawTextTokens.length - 2 - edgeWindow1.sliceIndex }, + { text: '\'', index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex } + ], match: { + text: 'can\'', + index: rawTextTokens.length - 2 - edgeWindow1.sliceIndex + } + }], + splits: [], + unmappedEdits: [], + edgeWindow: edgeWindow1 + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: 't', deleteLeft: 0 }); + return map; + })() + }; + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.merges = []; + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '.', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: '.', deleteLeft: 0 }); + return map; + })() + + assert.deepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + assert.notEqual(key2, key1); + }); + + it("properly notes new boundary token length on boundary-final split", () => { + const rawTextTokens = ['she', ' ', 'says', ' ', 'I', ' ', 'can\'']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const edgeWindow1 = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '.', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [{ + // The indices specified here are the edge-window-internal indices. + input: { + text: 'can\'', + index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex + }, matches: [ + { text: 'can', index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex, textOffset: 0 }, + { text: '\'', index: rawTextTokens.length - 0 - edgeWindow1.sliceIndex, textOffset: 3 } + ] + }], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '.', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + // Creates a `'.` token. The two chars should technically be separate + // tokens, but... we should still get a distinct key if they're combined + // this way. + map.set(0, { insert: '.', deleteLeft: 0 }); + return map; + })() + }; + + const key1 = precomputationSubsetKeyer(precomputation1); + // Boundary token says is length 5 - as if appending to `can'`, not to just `'`. + assert.isFalse(key1.indexOf("BI@0-5") > -1, "The key's split marker length does not correspond to last split token length"); + // Boundary token says is length 2 - as if appending to just `'`, not to `can'`. + assert.isTrue(key1.indexOf("BI@0-2") > -1); + }); + + it("generates different keys for matching transforms when one causes token split", () => { + const rawTextTokens = ['she', ' ', 'says', ' ', 'I', ' ', 'can\'']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const edgeWindow1 = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '.', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [{ + // The indices specified here are the edge-window-internal indices. + input: { + text: 'can\'', + index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex + }, matches: [ + { text: 'can', index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex, textOffset: 0 }, + { text: '\'', index: rawTextTokens.length - 0 - edgeWindow1.sliceIndex, textOffset: 3 } + ] + }], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '.', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + // Creates a `'.` token. The two chars should technically be separate + // tokens, but... we should still get a distinct key if they're combined + // this way. + map.set(0, { insert: '.', deleteLeft: 0 }); + return map; + })() + }; + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.splits = []; + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 't', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: 't', deleteLeft: 0 }); + return map; + })() + + assert.deepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + assert.notEqual(key2, key1); + }); +}); \ No newline at end of file