From db5e77c0dc896a4e15bde945f38384da79906985 Mon Sep 17 00:00:00 2001 From: Joshua Horton Date: Tue, 9 Sep 2025 10:02:04 -0500 Subject: [PATCH 1/3] refactor(web): move traversal-less correction code to own helper --- .../worker-thread/src/main/predict-helpers.ts | 126 +++++++++++------- 1 file changed, 76 insertions(+), 50 deletions(-) diff --git a/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts b/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts index 43fd2deb02..d325d12fd1 100644 --- a/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts +++ b/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts @@ -151,6 +151,77 @@ export function tupleDisplayOrderSort(a: CorrectionPredictionTuple, b: Correctio return b.totalProb - a.totalProb; } +export async function correctAndEnumerateWithoutTraversals( + lexicalModel: LexicalModel, + transformDistribution: Distribution, + context: Context +): Promise<{ + /** + * For models that support correction-search caching, this provides the + * cached object corresponding to this method's operation. + * + * Otherwise, is `null`. + */ + postContextState?: ContextState; + + /** + * The suggestions generated based on the user's input state. + */ + rawPredictions: CorrectionPredictionTuple[]; + + /** + * The id of a prior ContextTransition event that triggered a Suggestion found + * at the end of the Context. Will be undefined if no edits have occurred + * since the Suggestion was applied. + */ + revertableTransitionId?: number +}> { + const inputTransform = transformDistribution[0].sample; + let rawPredictions: CorrectionPredictionTuple[] = []; + + let predictionRoots: ProbabilityMass[]; + + // Only allow new-word suggestions if space was the most likely keypress. + const allowSpace = TransformUtils.isWhitespace(inputTransform); + const allowBksp = TransformUtils.isBackspace(inputTransform); + + // Generates raw prediction distributions for each valid input. Can only 'correct' + // against the final input. + // + // This is the old, 12.0-13.0 'correction' style. + if(allowSpace) { + // Detect start of new word; prevent whitespace loss here. + predictionRoots = [{sample: inputTransform, p: 1.0}]; + } else { + predictionRoots = transformDistribution.map((alt) => { + let transform = alt.sample; + + // Filter out special keys unless they're expected. + if(TransformUtils.isWhitespace(transform) && !allowSpace) { + return null; + } else if(TransformUtils.isBackspace(transform) && !allowBksp) { + return null; + } + + return alt; + }); + } + + // Remove `null` entries. + predictionRoots = predictionRoots.filter(tuple => !!tuple); + + // Running in bulk over all suggestions, duplicate entries may be possible. + rawPredictions = predictFromCorrections(lexicalModel, predictionRoots, context); + if(allowSpace) { + rawPredictions.forEach((entry) => entry.preservationTransform = inputTransform); + } + + return { + postContextState: null, + rawPredictions: rawPredictions + }; +} + /** * This method performs the correction-search and model-lookup operations for * prediction generation by using the user's context state and potential @@ -186,14 +257,6 @@ export async function correctAndEnumerate( */ revertableTransitionId?: number }> { - const wordbreak = determineModelWordbreaker(lexicalModel); - - // Assertion / pre-condition: `transformDistribution` should be sorted! - const inputTransform = transformDistribution[0].sample; - - const postContext = models.applyTransform(inputTransform, context); - let rawPredictions: CorrectionPredictionTuple[] = []; - // If `this.contextTracker` does not exist, we don't have the // `LexiconTraversal` pattern available to us. We're unable to efficiently // iterate through the lexicon as a result, so we use a far lazier pattern - @@ -202,47 +265,7 @@ export async function correctAndEnumerate( // It's mostly here to support models compiled before Keyman 14.0, which was // when the `LexiconTraversal` pattern was established. if(!contextTracker) { - let predictionRoots: ProbabilityMass[]; - - // Only allow new-word suggestions if space was the most likely keypress. - const allowSpace = TransformUtils.isWhitespace(inputTransform); - const allowBksp = TransformUtils.isBackspace(inputTransform); - - // Generates raw prediction distributions for each valid input. Can only 'correct' - // against the final input. - // - // This is the old, 12.0-13.0 'correction' style. - if(allowSpace) { - // Detect start of new word; prevent whitespace loss here. - predictionRoots = [{sample: inputTransform, p: 1.0}]; - } else { - predictionRoots = transformDistribution.map((alt) => { - let transform = alt.sample; - - // Filter out special keys unless they're expected. - if(TransformUtils.isWhitespace(transform) && !allowSpace) { - return null; - } else if(TransformUtils.isBackspace(transform) && !allowBksp) { - return null; - } - - return alt; - }); - } - - // Remove `null` entries. - predictionRoots = predictionRoots.filter(tuple => !!tuple); - - // Running in bulk over all suggestions, duplicate entries may be possible. - rawPredictions = predictFromCorrections(lexicalModel, predictionRoots, context); - if(allowSpace) { - rawPredictions.forEach((entry) => entry.preservationTransform = inputTransform); - } - - return { - postContextState: null, - rawPredictions: rawPredictions - }; + return correctAndEnumerateWithoutTraversals(lexicalModel, transformDistribution, context); } // 'else': the current, 14.0+ pattern, which is able to leverage @@ -267,7 +290,6 @@ export async function correctAndEnumerate( // If both are smaller than the threshold, the text should match exactly. return a.left == b.left; } - }; const lastTransition = contextTracker.latest; @@ -280,6 +302,7 @@ export async function correctAndEnumerate( contextState = lastTransition.base; } + const inputTransform = transformDistribution[0].sample; if(!contextState){ console.warn("Unexpected context state occurred as prediction base context"); contextTracker.reset(context, inputTransform.id); @@ -289,6 +312,7 @@ export async function correctAndEnumerate( // Corrections and predictions are based upon the post-context state, though. let transition = contextTracker.latest; const inputIsEmpty = TransformUtils.isEmpty(inputTransform) && transformDistribution.length == 1; + const postContext = models.applyTransform(inputTransform, context); // Don't replace any applied-suggestion data if we have a request to trigger with // the current context state. @@ -348,6 +372,7 @@ export async function correctAndEnumerate( const alignment = postContextState.tokenization.alignment; // If the context now has more tokens, the token we'll be 'predicting' didn't originally exist. + const wordbreak = determineModelWordbreaker(lexicalModel); if(transition.preservationTransform && alignment?.canAlign && alignment.tailTokenShift > 0) { // As the word/token being corrected/predicted didn't originally exist, there's no // part of it to 'replace'. (Suggestions are applied to the pre-transform state.) @@ -423,6 +448,7 @@ export async function correctAndEnumerate( } // Only run the correction search when corrections are enabled. + let rawPredictions: CorrectionPredictionTuple[] = []; for await(let match of searchSpace.getBestMatches(timer)) { // Corrections obtained: now to predict from them! const correction = match.matchString; From eb679488a4e5db04fc29ba6dceafcaf517e5d789 Mon Sep 17 00:00:00 2001 From: Joshua Horton Date: Tue, 9 Sep 2025 10:45:06 -0500 Subject: [PATCH 2/3] refactor(web): spin off matchBaseContextState, determineContextTransition as helpers --- .../worker-thread/src/main/predict-helpers.ts | 183 +++++++++++------- 1 file changed, 117 insertions(+), 66 deletions(-) diff --git a/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts b/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts index d325d12fd1..cb39385fc2 100644 --- a/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts +++ b/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts @@ -222,6 +222,119 @@ export async function correctAndEnumerateWithoutTraversals( }; } +/** + * Determines the most recent ContextState corresponding to the incoming + * Context, assuming no context-reset operations have occurred. Their contents + * may not match perfectly, but they should be alignable with no edits not + * caused by the sliding context window. + * + * If no contexts align, this will trigger a warning and a context reset. + * @param contextTracker The cache for previously-analyzed context states + * @param context The incoming context + * @param inputTransform The ID for the incoming context transition (only used + * if a context reset is necessary) + * @returns + */ +export function matchBaseContextState( + contextTracker: ContextTracker, + context: Context, + transitionId: number +): ContextState { + const lengthThreshold = contextTracker.configuration?.leftContextCodePoints ?? Number.POSITIVE_INFINITY; + + const contextsMatch = (a: Context, b: Context) => { + // If either context's window is equal to or greater than the threshold + // length, there's a good chance something slid into or out of range. + if(a.left.length >= lengthThreshold || b.left.length >= lengthThreshold) { + return isSubstitutionAlignable(a.left, b.left); + } else { + // If both are smaller than the threshold, the text should match exactly. + return a.left == b.left; + } + }; + + const lastTransition = contextTracker.latest; + let contextState: ContextState; + + // Note that the "final" context from the last operation will have any + // characters substituted - only insert (if context window was shortened) or + // delete (if lengthened). No substitutions possible, as no rules will have + // occurred - the ONLY change is the amount of known text vs the context + // window's range. + if(contextsMatch(lastTransition.final.context, context)) { + contextState = lastTransition.final; + } else if(contextsMatch(lastTransition.base.context, context)) { + // Multitap case: if we reverted back to the original underlying context, + // rather than using the previous output context. + // + // This may also arise for text input that triggers auto-correct, as the + // incoming text should be processed after applying the suggestion, as + // applying the suggestion also appends the incoming text. + contextState = lastTransition.base; + } + + if(!contextState){ + console.warn("Unexpected context state occurred as prediction base context"); + contextTracker.reset(context, transitionId); + contextState = contextTracker.latest.base; + } + + return contextState; +} + +/** + * Determines the tokenization for the context after any incoming edits are + * applied. The tokenization(s) then determine(s) what word/token is the root + * for any corrections or predictions to be generated. + * + * Any incoming fat-finger data is applied to its corresponding token(s) here. + * @param contextTracker + * @param baseContextState + * @param context + * @param transformDistribution + * @returns + */ +export function determineContextTransition( + contextTracker: ContextTracker, + baseContextState: ContextState, + context: Context, + transformDistribution: Distribution +): ContextTransition { + const inputTransform = transformDistribution[0].sample; + + // Corrections and predictions are based upon the post-context state, though. -- MOVE THIS COMMENT! ** + let transition = contextTracker.latest; + const inputIsEmpty = TransformUtils.isEmpty(inputTransform) && transformDistribution.length == 1; + const postContext = models.applyTransform(inputTransform, context); + + // Don't replace any applied-suggestion data if we have a request to trigger with + // the current context state. + if(inputIsEmpty) { + // Directly build a simple empty transition that duplicates the last seen state. + const priorState = contextTracker.latest.final; + transition = new ContextTransition(priorState, inputTransform.id); + transition.finalize(priorState, transformDistribution); + } else if( + // If the input matches something we've already seen (say, a ' ' or '.' + // that auto-applied a suggestion). + transition.transitionId == inputTransform.id && + transition.final.context.left == postContext.left + ) { + // Retrieve the already-performed transition and abort. + transition.inputDistribution = transformDistribution; + return transition; + } else { + transition = baseContextState.analyzeTransition(context, transformDistribution); + if(!transition) { + console.warn("Unexpected failure when computing context-state transition"); + // Only known remaining use of `analyzeState` currently - and it's as a failsafe! + transition = contextTracker.analyzeState(contextTracker.model, context, transformDistribution); + } + } + + return transition; +} + /** * This method performs the correction-search and model-lookup operations for * prediction generation by using the user's context state and potential @@ -271,67 +384,15 @@ export async function correctAndEnumerate( // 'else': the current, 14.0+ pattern, which is able to leverage // `LexiconTraversal`s for greater search efficiency. This pattern // facilitates a more thorough correction-search pattern. - - // Token replacement benefits greatly from knowledge of the prior context state. - // Note: not long-term viable - context is a sliding window, after all! - // What we actually need is a sliding-window comparison. - // - // Note that the "final" context from the last operation will not substitute any - // characters - only insert (if context window was shortened) or delete (if - // lengthened). No substitutions possible, as no rules will have occurred - the - // ONLY change is the amount of known text vs the context window's range. - const lengthThreshold = contextTracker.configuration?.leftContextCodePoints ?? Number.POSITIVE_INFINITY; - const contextsMatch = (a: Context, b: Context) => { - // If either context's window is equal to or greater than the threshold length, there's a good - // chance something slid into or out of range. - if(a.left.length >= lengthThreshold || b.left.length >= lengthThreshold) { - return isSubstitutionAlignable(a.left, b.left); - } else { - // If both are smaller than the threshold, the text should match exactly. - return a.left == b.left; - } - }; - - const lastTransition = contextTracker.latest; - let contextState: ContextState; - if(contextsMatch(lastTransition.final.context, context)) { - contextState = lastTransition.final; - } else if(contextsMatch(lastTransition.base.context, context)) { - // Multitap case: if we reverted back to the original underlying context, - // rather than using the previous output context. - contextState = lastTransition.base; - } - const inputTransform = transformDistribution[0].sample; - if(!contextState){ - console.warn("Unexpected context state occurred as prediction base context"); - contextTracker.reset(context, inputTransform.id); - contextState = contextTracker.latest.base; - } + let contextState = matchBaseContextState(contextTracker, context, inputTransform.id); // Corrections and predictions are based upon the post-context state, though. - let transition = contextTracker.latest; - const inputIsEmpty = TransformUtils.isEmpty(inputTransform) && transformDistribution.length == 1; + const baseTransition = contextTracker.latest; const postContext = models.applyTransform(inputTransform, context); - // Don't replace any applied-suggestion data if we have a request to trigger with - // the current context state. - if(inputIsEmpty) { - // Directly build a simple empty transition that duplicates the last seen state. - const priorState = contextTracker.latest.final; - transition = new ContextTransition(priorState, inputTransform.id); - transition.finalize(priorState, transformDistribution); - } else if( - // If the input matches something we've already seen (say, a ' ' or '.' - // that auto-applied a suggestion). - transition.transitionId == inputTransform.id && - transition.final.context.left == postContext.left - ) { - // Retrieve the already-performed transition and re-predict based - // on its values. - transition.inputDistribution = transformDistribution; - contextState = transition.final; - + const transition = determineContextTransition(contextTracker, contextState, context, transformDistribution); + if(transition == baseTransition) { // Not yet done; we may want to consider saving the fat-finger distribution of // incoming text for this case for correction-search - we didn't get it // when applying a prior suggestion. @@ -340,13 +401,6 @@ export async function correctAndEnumerate( return { rawPredictions: [], postContextState: transition.final - }; - } else { - transition = contextState.analyzeTransition(context, transformDistribution); - if(!transition) { - console.warn("Unexpected failure when computing context-state transition"); - // Only known remaining use of `analyzeState` currently - and it's as a failsafe! - transition = contextTracker.analyzeState(lexicalModel, context, transformDistribution); } } @@ -356,9 +410,6 @@ export async function correctAndEnumerate( // TODO: Should we filter backspaces & whitespaces out of the transform distribution? // Ideally, the answer (in the future) will be no, but leaving it in right now may pose an issue. - // Rather than go "full hog" and make a priority queue out of the eventual, future competing search spaces... - // let's just note that right now, there will only ever be one. - // // The 'eventual' logic will be significantly more complex, though still manageable. const searchSpace = postContextState.tokenization.tail.searchSpace; From 439b9c3f016c4dff28e53c9ce7f4895a4e4f6e8a Mon Sep 17 00:00:00 2001 From: Joshua Horton Date: Tue, 9 Sep 2025 11:29:16 -0500 Subject: [PATCH 3/3] refactor(web): spin off determineSuggestionAlignment method --- .../worker-thread/src/main/predict-helpers.ts | 207 +++++++++--------- 1 file changed, 109 insertions(+), 98 deletions(-) diff --git a/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts b/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts index cb39385fc2..42ca17189f 100644 --- a/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts +++ b/web/src/engine/predictive-text/worker-thread/src/main/predict-helpers.ts @@ -302,7 +302,6 @@ export function determineContextTransition( ): ContextTransition { const inputTransform = transformDistribution[0].sample; - // Corrections and predictions are based upon the post-context state, though. -- MOVE THIS COMMENT! ** let transition = contextTracker.latest; const inputIsEmpty = TransformUtils.isEmpty(inputTransform) && transformDistribution.length == 1; const postContext = models.applyTransform(inputTransform, context); @@ -332,9 +331,78 @@ export function determineContextTransition( } } + contextTracker.latest = transition; return transition; } +/** + * Determines where the context for prediction-generation should be rooted and how + * much of the context it should replace. + * @param transition + * @param lexicalModel + * @returns + */ +export function determineSuggestionAlignment( + transition: ContextTransition, + lexicalModel: LexicalModel +): { + /** + * The context to use directly for generating predictions from the model. + */ + predictionContext: Context, + /** + * The total number of characters to delete for generated suggestions + * in order to replace the prediction root token entirely. + */ + deleteLeft: number +} { + const alignment = transition.final.tokenization.alignment; + const context = transition.base.context; + const postContext = transition.final.context; + const inputTransform = transition.inputDistribution[0].sample; + let deleteLeft: number; + + // If the context now has more tokens, the token we'll be 'predicting' didn't originally exist. + const wordbreak = determineModelWordbreaker(lexicalModel); + + // Is the token under construction newly-constructed / is there no pre-existing root? + if(transition.preservationTransform && alignment?.canAlign && alignment.tailTokenShift > 0) { + return { + // If the new token is due to whitespace or due to a different input type + // that would likely imply a tokenization boundary, infer 'new word' mode. + // Apply any part of the context change that is not considered to be up + // for correction. + predictionContext: models.applyTransform(transition.preservationTransform, context), + // As the word/token being corrected/predicted didn't originally exist, + // there's no part of it to 'replace'. (Suggestions are applied to the + // pre-transform state.) + deleteLeft: 0 + }; + // If the tokenized context length is shorter... sounds like a backspace (or similar). + } else if (alignment?.canAlign && alignment.tailTokenShift < 0) { + /* Ooh, we've dropped context here. Almost certainly from a backspace or + * similar effect. Even if we drop multiple tokens... well, we know exactly + * how many chars were actually deleted - `inputTransform.deleteLeft`. Since + * we replace a word being corrected/predicted, we take length of the + * remaining context's tail token in addition to however far was deleted to + * reach that state. + */ + deleteLeft = KMWString.length(wordbreak(postContext)) + inputTransform.deleteLeft; + } else { + // Suggestions are applied to the pre-input context, so get the token's original length. + // We're on the same token, so just delete its text for the replacement op. + deleteLeft = KMWString.length(wordbreak(context)); + } + + // Did the wordbreaker (or similar) append a blank token before the caret? If so, + // preserve that by preventing corrections from triggering left-deletion. + if(transition.final.tokenization.tail.exampleInput == '') { + deleteLeft = 0; + } + + return { predictionContext: context, deleteLeft }; +} + /** * This method performs the correction-search and model-lookup operations for * prediction generation by using the user's context state and potential @@ -389,8 +457,6 @@ export async function correctAndEnumerate( // Corrections and predictions are based upon the post-context state, though. const baseTransition = contextTracker.latest; - const postContext = models.applyTransform(inputTransform, context); - const transition = determineContextTransition(contextTracker, contextState, context, transformDistribution); if(transition == baseTransition) { // Not yet done; we may want to consider saving the fat-finger distribution of @@ -404,95 +470,38 @@ export async function correctAndEnumerate( } } - contextTracker.latest = transition; - const postContextState = transition.final; + // No matter the prediction, once we know the root of the prediction, we'll always 'replace' the + // same amount of text. We can handle this before the big 'prediction root' loop. + const { predictionContext: predictionContext, deleteLeft } = determineSuggestionAlignment(transition, lexicalModel); // TODO: Should we filter backspaces & whitespaces out of the transform distribution? // Ideally, the answer (in the future) will be no, but leaving it in right now may pose an issue. // The 'eventual' logic will be significantly more complex, though still manageable. - const searchSpace = postContextState.tokenization.tail.searchSpace; + const searchSpace = transition.final.tokenization.tail.searchSpace; - // No matter the prediction, once we know the root of the prediction, we'll always 'replace' the - // same amount of text. We can handle this before the big 'prediction root' loop. - let deleteLeft = 0; - - // The amount of text to 'replace' depends upon whatever sort of context change occurs - // from the received input. - const postContextTokens = postContextState.tokenization.tokens; - const alignment = postContextState.tokenization.alignment; - - // If the context now has more tokens, the token we'll be 'predicting' didn't originally exist. - const wordbreak = determineModelWordbreaker(lexicalModel); - if(transition.preservationTransform && alignment?.canAlign && alignment.tailTokenShift > 0) { - // As the word/token being corrected/predicted didn't originally exist, there's no - // part of it to 'replace'. (Suggestions are applied to the pre-transform state.) - deleteLeft = 0; - - // If the new token is due to whitespace or due to a different input type that would - // likely imply a tokenization boundary, infer 'new word' mode. - // Apply any part of the context change that is not considered - // to be up for correction. - context = models.applyTransform(transition.preservationTransform, context); - // If the tokenized context length is shorter... sounds like a backspace (or similar). - } else if (alignment?.canAlign && alignment.tailTokenShift < 0) { - // TODO: may need adjustment / refactoring for complex, word-boundary crossing transforms - // and easier unit testing of this logic! - - /* Ooh, we've dropped context here. Almost certainly from a backspace. - * Even if we drop multiple tokens... well, we know exactly how many chars - * were actually deleted - `inputTransform.deleteLeft`. - * Since we replace a word being corrected/predicted, we take length of the remaining - * context's tail token in addition to however far was deleted to reach that state. - */ - deleteLeft = KMWString.length(wordbreak(postContext)) + inputTransform.deleteLeft; - } else { - // Suggestions are applied to the pre-input context, so get the token's original length. - // We're on the same token, so just delete its text for the replacement op. - deleteLeft = KMWString.length(wordbreak(context)); - } - - // Is the token under construction newly-constructed / is there no pre-existing root? - // If so, we want to strongly avoid overcorrection, even for 'nearby' keys. - // (Strong lexical frequency differences can easily cause overcorrection when only - // one key's available.) + // If corrections are not enabled, bypass the correction search aspect + // entirely. No need to 'search' - just do a direct lookup. // - // NOTE: we only want this applied word-initially, when any corrections 'correct' - // 100% of the word. Things are generally fine once it's not "all or nothing." - let tailToken = postContextTokens[postContextTokens.length - 1]; - - // Did the wordbreaker (or similar) append a blank token before the caret? If so, - // preserve that by preventing corrections from triggering left-deletion. - if(tailToken.exampleInput == '') { - deleteLeft = 0; - } - - const isTokenStart = tailToken.searchSpace.inputSequence.length <= 1; - - // TODO: whitespace, backspace filtering. Do it here. - // Whitespace is probably fine, actually. Less sure about backspace. - - let bestCorrectionCost: number; - let correctionPredictionMap: Record> = {}; - - // If corrections are not enabled, bypass the correction search aspect entirely. - // No need to 'search' - just do a direct lookup. + // To be clear: this IS how we actually tell that corrections are disabled - + // when no fat-finger data is available. if(!searchSpace.correctionsEnabled) { + const wordbreak = determineModelWordbreaker(lexicalModel); const predictionRoot = { sample: { - insert: wordbreak(postContext), // insert correction string + insert: wordbreak(transition.final.context), deleteLeft: deleteLeft, id: inputTransform.id // The correction should always be based on the most recent external transform/transcription ID. }, p: 1.0 }; - let predictions = predictFromCorrections(lexicalModel, [predictionRoot], context); + const predictions = predictFromCorrections(lexicalModel, [predictionRoot], predictionContext); predictions.forEach((entry) => entry.preservationTransform = transition.preservationTransform); // Only one 'correction' / prediction root is allowed - the actual text. return { - postContextState: postContextState, + postContextState: transition.final, rawPredictions: predictions, revertableTransitionId: transition.revertableTransitionId } @@ -500,7 +509,9 @@ export async function correctAndEnumerate( // Only run the correction search when corrections are enabled. let rawPredictions: CorrectionPredictionTuple[] = []; - for await(let match of searchSpace.getBestMatches(timer)) { + let bestCorrectionCost: number; + const correctionPredictionMap: Record> = {}; + for await(const match of searchSpace.getBestMatches(timer)) { // Corrections obtained: now to predict from them! const correction = match.matchString; @@ -527,30 +538,30 @@ export async function correctAndEnumerate( let rootCost = match.totalCost; /* If we're dealing with the FIRST keystroke of a new sequence, we'll **dramatically** boost - * the exponent to ensure only VERY nearby corrections have a chance of winning, and only if - * there are significantly more likely words. We only need this to allow very minor fat-finger - * adjustments for 100% keystroke-sequence corrections in order to prevent finickiness on - * key borders. - * - * Technically, the probabilities this produces won't be normalized as-is... but there's no - * true NEED to do so for it, even if it'd be 'nice to have'. Consistently tracking when - * to apply it could become tricky, so it's simpler to leave out. - * - * Worst-case, it's possible to temporarily add normalization if a code deep-dive - * is needed in the future. - */ - if(isTokenStart) { + * the exponent to ensure only VERY nearby corrections have a chance of winning, and only if + * there are significantly more likely words. We only need this to allow very minor fat-finger + * adjustments for 100% keystroke-sequence corrections in order to prevent finickiness on + * key borders. + * + * Technically, the probabilities this produces won't be normalized as-is... but there's no + * true NEED to do so for it, even if it'd be 'nice to have'. Consistently tracking when + * to apply it could become tricky, so it's simpler to leave out. + * + * Worst-case, it's possible to temporarily add normalization if a code deep-dive + * is needed in the future. + */ + if(searchSpace.inputSequence.length <= 1) { /* Suppose a key distribution: most likely with p=0.5, second-most with 0.4 - a pretty - * ambiguous case that would only arise very near the center of the boundary between two keys. - * Raising (0.5/0.4)^16 ~= 35.53. (At time of writing, SINGLE_CHAR_KEY_PROB_EXPONENT = 16.) - * That seems 'within reason' for correction very near boundaries. - * - * So, with the second-most-likely key being that close in probability, its best suggestion - * must be ~ 35.5x more likely than that of the truly-most-likely key to "win". So, it's not - * a HARD cutoff, but more of a 'soft' one. Keeping the principles in mind documented above, - * it's possible to tweak this to a more harsh or lenient setting if desired, rather than - * being totally "all or nothing" on which key is taken for highly-ambiguous keypresses. - */ + * ambiguous case that would only arise very near the center of the boundary between two keys. + * Raising (0.5/0.4)^16 ~= 35.53. (At time of writing, SINGLE_CHAR_KEY_PROB_EXPONENT = 16.) + * That seems 'within reason' for correction very near boundaries. + * + * So, with the second-most-likely key being that close in probability, its best suggestion + * must be ~ 35.5x more likely than that of the truly-most-likely key to "win". So, it's not + * a HARD cutoff, but more of a 'soft' one. Keeping the principles in mind documented above, + * it's possible to tweak this to a more harsh or lenient setting if desired, rather than + * being totally "all or nothing" on which key is taken for highly-ambiguous keypresses. + */ rootCost *= ModelCompositor.SINGLE_CHAR_KEY_PROB_EXPONENT; // note the `Math.exp` below. } @@ -559,7 +570,7 @@ export async function correctAndEnumerate( p: Math.exp(-rootCost) }; - let predictions = predictFromCorrections(lexicalModel, [predictionRoot], context); + let predictions = predictFromCorrections(lexicalModel, [predictionRoot], predictionContext); predictions.forEach((entry) => entry.preservationTransform = transition.preservationTransform); // Only set 'best correction' cost when a correction ACTUALLY YIELDS predictions. @@ -586,7 +597,7 @@ export async function correctAndEnumerate( // console.log(`execute: ${timer.executionTime}, deferred: ${timer.deferredTime}`); //, total since start: ${timer.timeSinceConstruction}`); return { - postContextState: postContextState, + postContextState: transition.final, rawPredictions: rawPredictions, revertableTransitionId: transition.revertableTransitionId };