diff --git a/web/src/engine/predictive-text/worker-thread/src/main/correction/tokenization-subsets.ts b/web/src/engine/predictive-text/worker-thread/src/main/correction/tokenization-subsets.ts new file mode 100644 index 0000000000..2e9304ece9 --- /dev/null +++ b/web/src/engine/predictive-text/worker-thread/src/main/correction/tokenization-subsets.ts @@ -0,0 +1,83 @@ +import { SENTINEL_CODE_UNIT } from '@keymanapp/models-templates'; +import { KMWString } from '@keymanapp/web-utils'; + +import { TokenizationTransitionEdits } from './context-tokenization.js'; + +export function precomputationSubsetKeyer(tokenizationEdits: TokenizationTransitionEdits): string { + const { alignment, tokenizedTransform } = tokenizationEdits; + const { edgeWindow, merges, splits, unmappedEdits } = alignment; + const components: string[] = []; + + // First entry: based on the edge window. The real key: what's the edit + // boundary? We need to apply to the same token and portion thereof. + const editBoundary = edgeWindow.editBoundary; + + // It's not about the boundary text - we just need to ensure it's the 'same' + // token - comprised of the same keystrokes. `sourceRangeKey` reflects the + // actual input for the source keystrokes. We might have deleted part of it + // in this tokenization, but that doesn't matter here - we want to imply the + // represented keystroke range. + const boundaryEdgeIndex = editBoundary.tokenIndex - edgeWindow.sliceIndex; + const boundaryComponent = `B${editBoundary.tokenIndex}=${editBoundary.sourceRangeKey}`; + + components.push(boundaryComponent); + + // Identify the new boundary token's length - as it appears after any related + // merges or splits. + let boundaryTextLen = KMWString.length(editBoundary.text); + const boundaryMerge = merges.find((m) => m.inputs.find(i => i.index == boundaryEdgeIndex)); + const boundarySplit = splits.find((s) => s.input.index == boundaryEdgeIndex); + if(boundaryMerge) { + boundaryTextLen = KMWString.length(boundaryMerge.match.text); + } else if(boundarySplit) { + boundaryTextLen = KMWString.length(boundarySplit.matches[boundarySplit.matches.length - 1].text); + } + + // Now, based on the transform tokenization. We want to force uniqueness for + // all variations of result length on each tokenized transform resulting from + // the precomputation's represented keystroke. + for(const {0: relativeIndex, 1: transform} of tokenizedTransform.entries()) { + const insertLen = KMWString.length(transform.insert); + if(relativeIndex > 0) { + // The true boundary lie before the insert if the value is non-zero; + // don't differentiate here! + boundaryTextLen = 0; + } + + if(boundaryTextLen) { + // transform.deleteLeft was already handled during boundary computation - + // do not include it here! + components.push(`BI@${relativeIndex}-${boundaryTextLen + insertLen}`); + boundaryTextLen = 0; + } else { + components.push(`I@${relativeIndex}-${insertLen}`); + } + } + + if(merges.length > 0) { + components.push('M:' + merges.map((matchMap) => { + // Text may be more unique, but is likely unnecessary; index yields shorter, + // easier to process keys. + const inputPortion = matchMap.inputs.map(i => '' + i.index).join('+'); + return `M:${inputPortion}=>${matchMap.match.index}`; + }).join(',')); + } + + if(splits.length > 0) { + components.push('S:' + splits.map((matchMap) => { + // Text may be more unique, but is likely unnecessary; index yields shorter, + // easier to process keys. + const matchPortion = matchMap.matches.map(m => '' + m.index).join('+'); + return `${matchMap.input.index}=>${matchPortion}`; + }).join(',')); + } + + if(unmappedEdits.length > 0) { + // We really shouldn't have these, let alone often. + components.push('UE:' + unmappedEdits.map((edit) => { + return `${edit.op}(${edit.input ?? ''}-${edit.match ?? ''}`; + }).join(',')); + } + + return components.join(SENTINEL_CODE_UNIT); +} diff --git a/web/src/engine/predictive-text/worker-thread/src/main/test-index.ts b/web/src/engine/predictive-text/worker-thread/src/main/test-index.ts index b17e9c598d..2c64db440b 100644 --- a/web/src/engine/predictive-text/worker-thread/src/main/test-index.ts +++ b/web/src/engine/predictive-text/worker-thread/src/main/test-index.ts @@ -6,6 +6,7 @@ export { ContextTracker } from './correction/context-tracker.js'; export { ContextTransition } from './correction/context-transition.js'; export * from './correction/alignment-helpers.js'; export { ExtendedEditOperation, SegmentableDistanceCalculation } from './correction/segmentable-calculation.js'; +export * from './correction/tokenization-subsets.js'; export * as correction from './correction/index.js'; export * from './model-helpers.js'; export * as models from './models/index.js'; diff --git a/web/src/test/auto/headless/engine/predictive-text/worker-thread/context/tokenization-subsets.tests.ts b/web/src/test/auto/headless/engine/predictive-text/worker-thread/context/tokenization-subsets.tests.ts new file mode 100644 index 0000000000..132117812e --- /dev/null +++ b/web/src/test/auto/headless/engine/predictive-text/worker-thread/context/tokenization-subsets.tests.ts @@ -0,0 +1,513 @@ +/* + * Keyman is copyright (C) SIL Global. MIT License. + * + * Created by jahorton on 2025-09-23 + * + * This file contains low-level tests designed to validate the behavior of the + * of the ContextTokenization class and its integration with the lower-level + * classes that it utilizes. + */ + +import { assert } from 'chai'; + +import { default as defaultBreaker } from '@keymanapp/models-wordbreakers'; +import { LexicalModelTypes } from '@keymanapp/common-types'; +import { deepCopy } from '@keymanapp/web-utils'; +import { jsonFixture } from '@keymanapp/common-test-resources/model-helpers.mjs'; + +import { buildEdgeWindow, ContextToken, ContextTokenization, models, precomputationSubsetKeyer, TokenizationTransitionEdits } from '@keymanapp/lm-worker/test-index'; + +import Transform = LexicalModelTypes.Transform; +import TrieModel = models.TrieModel; + +var plainModel = new TrieModel(jsonFixture('models/tries/english-1000'), + {wordBreaker: defaultBreaker}); + +function toToken(text: string) { + let isWhitespace = text == ' '; + let token = new ContextToken(plainModel, text); + token.isWhitespace = isWhitespace; + return token; +} + +describe('precomputationSubsetKeyer', function() { + it("safely generates keys for empty transition + empty contexts", () => { + const rawTextTokens = ['']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: '', deleteLeft: 0 }); + return map; + })() + }; + const key = precomputationSubsetKeyer(precomputation1); + + assert.isOk(key); + }); + + it("generates different keys for transforms of different insert lengths on the same context", () => { + const rawTextTokens = ['an', ' ', 'apple', ' ', 'a', ' ', 'day']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: '', deleteLeft: 0 }); + return map; + })() + }; + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 's', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: 's', deleteLeft: 0 }); + return map; + })() + + assert.deepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + assert.notEqual(key2, key1); + }); + + it("generates different keys for transforms of different deleteLeft lengths on the same context", () => { + const rawTextTokens = ['an', ' ', 'apple', ' ', 'a', ' ', 'day']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 's', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: 's', deleteLeft: 0 }); + return map; + })() + }; + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 'b', deleteLeft: 1, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: 'b', deleteLeft: 1 }); + return map; + })() + + assert.notDeepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + assert.notEqual(key2, key1); + }); + + it("generates matching keys when boundary token + transform results in equal length token with same source text (1)", () => { + const rawTextTokens = ['an', ' ', 'apple', ' ', 'a', ' ']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + // concept: inputs were 'd', 'a', 'te', 's' + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + [...tokenization.tokens, (() => { + const token = new ContextToken(plainModel, 'da'); + // source text: 'date' + token.addInput( + {trueTransform: {insert: 'te', deleteLeft: 0}, inputStartIndex: 0}, + [{sample: {insert: 'te', deleteLeft: 0}, p: 1}] + ); + return token; + })()], + { insert: 's', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: 's', deleteLeft: 0 }); + return map; + })() + }; + + // concept: inputs were 'd', 'a', 't', 'es' + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + [...tokenization.tokens, (() => { + const token = new ContextToken(plainModel, 'da'); + // source text: 'date' + token.addInput( + {trueTransform: {insert: 'te', deleteLeft: 0}, inputStartIndex: 0}, + [{sample: {insert: 't', deleteLeft: 0}, p: 1}] + ); + return token; + })()], + { insert: 'es', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: 'es', deleteLeft: 0 }); + return map; + })() + + assert.notDeepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + assert.equal(key2, key1); + }); + + // - also due to deleteLeft effects: 2 + 1 vs a 2 + 2-dl:1 + it("generates matching keys when boundary token + transform results in equal length token with same source text (1)", () => { + const rawTextTokens = ['an', ' ', 'apple', ' ', 'a', ' ']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + // concept: inputs were 'd', 'a', 'ts', 'e' (with delete-left 1) + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + [...tokenization.tokens, (() => { + const token = new ContextToken(plainModel, 'da'); + token.isPartial = true; + // source text: 'dat' + token.addInput( + {trueTransform: {insert: 't', deleteLeft: 0}, inputStartIndex: 0}, + [{sample: {insert: 'ts', deleteLeft: 0}, p: 1}] + ); + return token; + })()], + { insert: 'e', deleteLeft: 1, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: 'e', deleteLeft: 1 }); + return map; + })() + }; + + // concept: inputs were 'd', 'a', 't', 'e' + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + [...tokenization.tokens, (() => { + const token = new ContextToken(plainModel, 'da'); + token.isPartial = true; + // source text: 'dat' + token.addInput( + {trueTransform: {insert: 't', deleteLeft: 0}, inputStartIndex: 0}, + [{sample: {insert: 't', deleteLeft: 0}, p: 1}] + ); + return token; + })()], + { insert: 'e', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: 'e', deleteLeft: 0 }); + return map; + })() + + // delete lengths differ, but that should be it. + assert.notDeepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const alteredEdgeWindow = { + ...precomputation2.alignment.edgeWindow, + deleteLengths: [1] + } + assert.deepEqual(alteredEdgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + // Both result in the same net length of token (4) starting at the same + // point in the context / at the same keystroke. + assert.equal(key2, key1); + }); + + it("properly notes new boundary token length on boundary-final merge", () => { + const rawTextTokens = ['she', ' ', 'says', ' ', 'I', ' ', 'can', '\'']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const edgeWindow1 = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 't', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [{ + // The indices specified here are the edge-window-internal indices. + inputs: [ + { text: 'can', index: rawTextTokens.length - 2 - edgeWindow1.sliceIndex }, + { text: '\'', index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex } + ], match: { + text: 'can\'', + index: rawTextTokens.length - 2 - edgeWindow1.sliceIndex + } + }], + splits: [], + unmappedEdits: [], + edgeWindow: edgeWindow1 + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: 't', deleteLeft: 0 }); + return map; + })() + }; + + const key1 = precomputationSubsetKeyer(precomputation1); + // Boundary token says is length 2 - as if appending to just `'`, not to `can'`. + assert.isFalse(key1.indexOf("BI@0-2") > -1, "The key's merge marker length does not correspond to merged token length"); + // Boundary token says is length 5 - as if appending to `can'`. + assert.isTrue(key1.indexOf("BI@0-5") > -1); + }); + + it("generates different keys for matching transforms when one causes token merge", () => { + const rawTextTokens = ['she', ' ', 'says', ' ', 'I', ' ', 'can', '\'']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const edgeWindow1 = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 't', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [{ + // The indices specified here are the edge-window-internal indices. + inputs: [ + { text: 'can', index: rawTextTokens.length - 2 - edgeWindow1.sliceIndex }, + { text: '\'', index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex } + ], match: { + text: 'can\'', + index: rawTextTokens.length - 2 - edgeWindow1.sliceIndex + } + }], + splits: [], + unmappedEdits: [], + edgeWindow: edgeWindow1 + }, + tokenizedTransform: (() => { + const map = new Map(); + map.set(0, { insert: 't', deleteLeft: 0 }); + return map; + })() + }; + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.merges = []; + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '.', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: '.', deleteLeft: 0 }); + return map; + })() + + assert.deepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + assert.notEqual(key2, key1); + }); + + it("properly notes new boundary token length on boundary-final split", () => { + const rawTextTokens = ['she', ' ', 'says', ' ', 'I', ' ', 'can\'']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const edgeWindow1 = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '.', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [{ + // The indices specified here are the edge-window-internal indices. + input: { + text: 'can\'', + index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex + }, matches: [ + { text: 'can', index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex, textOffset: 0 }, + { text: '\'', index: rawTextTokens.length - 0 - edgeWindow1.sliceIndex, textOffset: 3 } + ] + }], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '.', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + // Creates a `'.` token. The two chars should technically be separate + // tokens, but... we should still get a distinct key if they're combined + // this way. + map.set(0, { insert: '.', deleteLeft: 0 }); + return map; + })() + }; + + const key1 = precomputationSubsetKeyer(precomputation1); + // Boundary token says is length 5 - as if appending to `can'`, not to just `'`. + assert.isFalse(key1.indexOf("BI@0-5") > -1, "The key's split marker length does not correspond to last split token length"); + // Boundary token says is length 2 - as if appending to just `'`, not to `can'`. + assert.isTrue(key1.indexOf("BI@0-2") > -1); + }); + + it("generates different keys for matching transforms when one causes token split", () => { + const rawTextTokens = ['she', ' ', 'says', ' ', 'I', ' ', 'can\'']; + let tokenization = new ContextTokenization(rawTextTokens.map((text => toToken(text)))); + + const edgeWindow1 = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '.', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + const precomputation1: TokenizationTransitionEdits = { + alignment: { + merges: [], + splits: [{ + // The indices specified here are the edge-window-internal indices. + input: { + text: 'can\'', + index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex + }, matches: [ + { text: 'can', index: rawTextTokens.length - 1 - edgeWindow1.sliceIndex, textOffset: 0 }, + { text: '\'', index: rawTextTokens.length - 0 - edgeWindow1.sliceIndex, textOffset: 3 } + ] + }], + unmappedEdits: [], + edgeWindow: { + ...buildEdgeWindow( + tokenization.tokens, + { insert: '.', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + } + }, + tokenizedTransform: (() => { + const map = new Map(); + // Creates a `'.` token. The two chars should technically be separate + // tokens, but... we should still get a distinct key if they're combined + // this way. + map.set(0, { insert: '.', deleteLeft: 0 }); + return map; + })() + }; + const precomputation2 = deepCopy(precomputation1); + precomputation2.alignment.splits = []; + precomputation2.alignment.edgeWindow = { + ...buildEdgeWindow( + tokenization.tokens, + { insert: 't', deleteLeft: 0, deleteRight: 0 }, + false + ), + retokenization: [...rawTextTokens] + }; + precomputation2.tokenizedTransform = (() => { + const map = new Map(); + map.set(0, { insert: 't', deleteLeft: 0 }); + return map; + })() + + assert.deepEqual(precomputation2.alignment.edgeWindow, precomputation1.alignment.edgeWindow); + const key1 = precomputationSubsetKeyer(precomputation1); + const key2 = precomputationSubsetKeyer(precomputation2); + + assert.notEqual(key2, key1); + }); +}); \ No newline at end of file