change(web): remove computeAlignment

While handy, this method did not adequately account for 'split' / 'merge' edit cases and also required special handling to avoid certain degenerate edit-path cases.  The newer method put in place (see #14874) handles such edits and avoids the degeneracy problem that resulted when relying on edit paths for the edited portion of context.

Build-bot: skip build:web
Test-bot: skip
This commit is contained in:
Joshua Horton 2025-10-02 13:14:24 -05:00
parent c6cf11afb4
commit 24e82f42ce
2 changed files with 2 additions and 1141 deletions

View file

@ -8,120 +8,7 @@
* edits for aligned context tokens.
*/
import { SENTINEL_CODE_UNIT } from '@keymanapp/models-templates';
import { ClassicalDistanceCalculation, computeDistance, EditTuple } from "./classical-calculation.js";
import { ExtendedEditOperation } from './segmentable-calculation.js';
/**
* Represents token-count values resulting from an alignment attempt between two
* different modeled context states.
*/
export type ContextStateAlignment = {
/**
* Denotes whether or not alignment is possible between two contexts.
*/
canAlign: false,
/**
* Indicates the edit path that could not be handled. (Useful for error reporting)
*
* The edit path does not include actual user text and is sanitized.
*/
editPath: EditTuple<ExtendedEditOperation>[];
} | {
/**
* Denotes whether or not alignment is possible between two contexts.
*/
canAlign: true,
/**
* Indicates the edit path that could not be handled. (Useful for error reporting)
*
* The edit path does not include actual user text and is sanitized.
*/
editPath: EditTuple<ExtendedEditOperation>[];
/**
* Notes the number of tokens added to the head of the 'incoming'/'new' context
* of the contexts being aligned. If negative, the incoming context deleted
* a token found in the 'original' / base context.
*
* For the alignment, [base context index] + leadTokenShift = [incoming context index].
*/
leadTokenShift: number,
/**
* Notes the number of tokens at the head of the 'incoming'/'new' context,
* perfectly aligned but edited for two successfully-alignable contexts. These
* tokens directly precede those that need no edits.
*
* When a token could be considered as either 'lead' or 'tail' edit, it will
* only be reported as a 'tail' edit.
*/
leadEditLength: number,
/**
* The count of tokens perfectly aligned, with no need for edits, for two successfully-
* alignable contexts.
*/
matchLength: number,
/**
* The count of tokens at the tail perfectly aligned (existing in both contexts) but
* edited for two successfully-alignable contexts. These tokens directly follow those
* that need no edits.
*/
tailEditLength: number,
/**
* The count of new tokens added at the end of the incoming context for two aligned contexts.
* If negative, the incoming context deleted a previously-existing token from the original.
*/
tailTokenShift: number
};
/**
* Determines the proper 'last match' index for a tokenized sequence based on its edit path.
*
* In particular, this method is designed to handle the following cases:
* - ['to', ' ', 'apple', ' ', ''] => ['to', ' ', 'apply', ' ', '']
* - ['to', ' ', 'apple', ' ', ''] => ['to', ' ', 'apply', ' ', 'n']
*
* Edit path for these example cases:
* - ['match', 'match', 'substitute', 'match', 'match']
* - ['match', 'match', 'substitute', 'match', 'substitute']
*
* In cases such as these, the late whitespace match should be considered 'edited'. While the
* ' ' is unedited, it follows the edited 'apple' => 'apply', so it must have been deleted and
* then re-inserted. As a result, the whitespace after 'to' is the true "last matched" token.
*
* Returns -1 if an unexpected edit other than 'substitute' occurs in the middle of the big
* 'match' block.
* @param editPath
* @returns
*/
export function getEditPathLastMatch(editPath: ExtendedEditOperation[], forAppliedSuggestion?: boolean) {
// Assertion: for a long context, the bulk of the edit path should be a
// continuous block of 'match' entries. If there's anything but a substitution
// in the middle, we have a context mismatch.
//
// That said, it is possible to apply a suggestion after a backspace. Anything
// after the substitution needs to be treated as a substitution rather than
// a match.
const firstMatch = editPath.indexOf('match');
const lastMatch = editPath.lastIndexOf('match');
if(firstMatch > -1) {
for(let i = firstMatch+1; i <= lastMatch; i++) {
if(editPath[i] != 'match') {
// fun case: ' ' + ' applied ' has an unusual edit path.
// we get an 'insert'.
return (
(editPath[i] == 'substitute')
|| (forAppliedSuggestion && editPath[i] == 'insert')
) ? (i - 1) : -1;
}
}
}
return lastMatch;
}
import { ClassicalDistanceCalculation, computeDistance } from "./classical-calculation.js";
/**
* Aligns two tokens on a character-by-character basis as needed for higher, token-level alignment
@ -214,229 +101,4 @@ export function isSubstitutionAlignable(
}
return true;
}
/**
* Determines the alignment between a new, incoming tokenization source and the
* tokenization modeled by the current instance.
* @param tokenizationToMatch Raw strings corresponding to the tokenization of the original context
* @param incomingTokenization Raw strings corresponding to the tokenization of the incoming context
* @param isSliding Notes if the context window is full (and sliding-alignment is particularly needed)
* @param forAppliedSuggestion When true, this asserts that the contexts are alignable and loosens
* alignment requirements accordingly.
* @returns Alignment data that details if and how the incoming tokenization aligns with
* the tokenization modeled by this instance.
*/
export function computeAlignment(
tokenizationToMatch: string[],
incomingTokenization: string[],
isSliding: boolean,
forAppliedSuggestion?: boolean
): ContextStateAlignment {
const src = tokenizationToMatch;
const dst = incomingTokenization;
// let changedEmptyTail = false;
if(dst[dst.length - 1] == '') {
// Only allow matching if the tokenizations are identical, thus the empty
// token was unaffected.
if(src.length != dst.length || src[dst.length - 1] != '') {
// Do not allow empty-token matches to match each other; this complicates
// things when applying zero-root suggestions.
//
// The SENTINEL char should never appear in raw text, thus should never
// match anything in the "tokenization to match".
dst[dst.length - 1] = SENTINEL_CODE_UNIT;
}
}
// Inverted order, since 'match' existed before our new context.
const mapping = computeDistance(
// Diagonal width allows asymmetric edits and is also needed to cover
// difference in length for the inputs. We should try to cover at least 2
// edits on one side in addition to potential length asymmetry.
new ClassicalDistanceCalculation({diagonalWidth: Math.abs(src.length - dst.length) + 3}),
src,
dst
);
// Later iteration: we could return this itself directly for use in alignment
// operations, rather than relying solely on the edit-op names.
let editPaths = mapping.editPath();
if(editPaths.length == 0) {
console.error(`Could not compute edit path for aligning contexts of length ${src.length}, ${dst.length}`);
}
let editPath = editPaths[0].map(t => t.op);
const failure: ContextStateAlignment = {
canAlign: false,
editPath: editPaths[0]
};
// Special case: new context bootstrapping - first token often substitutes.
// The text length is small enough that no words should be able to rotate out the start of the context.
// Special handling needed in case of no 'match'; the rest of the method assumes at least one 'match'.
if(editPath.length <= 3 && (editPath[0] == 'substitute' || editPath[0] == 'match')) {
let matchCount = 0;
let subCount = 0;
for(let i = 0; i < editPath.length; i++) {
if(editPath[i] == 'substitute') {
subCount++;
if(!forAppliedSuggestion && !isSubstitutionAlignable(incomingTokenization[i], tokenizationToMatch[i], true)) {
return failure;
}
} else if(editPath[i] == 'match') {
// If a substitution is already recorded, treat the 'match' as a substitution.
if(subCount > 0) {
subCount++;
} else {
matchCount++;
}
}
}
const insertCount = editPath.filter((entry) => entry == 'insert').length;
const deleteCount = editPath.filter((entry) => entry == 'delete').length;
return {
canAlign: true,
editPath: editPaths[0],
matchLength: matchCount,
leadTokenShift: 0,
leadEditLength: 0,
tailEditLength: subCount,
tailTokenShift: insertCount - deleteCount
}
}
// From here on assumes that at least one 'match' exists on the path.
// It all works great... once the context is long enough for at least one stable token.
const firstMatch = editPath.indexOf('match');
if(firstMatch == -1) {
// If there are no matches, there's no alignment.
return failure;
}
// Transpositions are not allowed at the token level during context alignment.
if(editPath.find((entry) => entry.indexOf('transpose') > -1)) {
return failure;
}
const lastMatch = getEditPathLastMatch(editPath, forAppliedSuggestion);
// Assertion: for a long context, the bulk of the edit path should be a
// continuous block of 'match' entries. If there's anything else in
// the middle, we have a context mismatch.
if(lastMatch == -1) {
return failure;
}
let matchLength = lastMatch - firstMatch + 1;
let tailInsertLength = 0;
let tailDeleteLength = 0;
for(let i = lastMatch; i < editPath.length; i++) {
if(editPath[i] == 'insert') {
tailInsertLength++;
} else if(editPath[i] == 'delete') {
tailDeleteLength++;
}
}
if(tailInsertLength > 0 && tailDeleteLength > 0) {
// Something's gone weird if this happens; that should appear as a substitution instead.
// Otherwise, we have a VERY niche edit scenario.
return failure;
}
const tailSubstituteLength = (editPath.length - 1 - lastMatch) - tailInsertLength - tailDeleteLength;
// If we have a perfect match with a pre-existing context, no mutations have
// happened; we have a 100% perfect match.
if(firstMatch == 0 && lastMatch == editPath.length - 1) {
return {
canAlign: true,
editPath: editPaths[0],
leadTokenShift: 0,
leadEditLength: 0,
matchLength,
tailEditLength: tailSubstituteLength,
tailTokenShift: tailInsertLength - tailDeleteLength
};
}
// The edit path calc tries to put substitutes first, before inserts.
// We don't want that on the leading edge.
const lastEarlyInsert = editPath.lastIndexOf('insert', firstMatch);
const firstSubstitute = editPath.indexOf('substitute');
if(firstSubstitute > -1 && firstSubstitute < firstMatch && firstSubstitute < lastEarlyInsert) {
editPath[firstSubstitute] = 'insert';
editPath[lastEarlyInsert] = 'substitute';
}
// If mutations HAVE happened, we need to double-check the context-state alignment.
let priorEdit: typeof editPath[0];
let leadTokensRemoved = 0;
let leadSubstitutions = 0;
// The `i` index below aligns based upon the index within the `tokenizationToMatch` sequence
// and how it would have to be edited to align to the `incomingTokenization` sequence.
for(let i = 0; i < firstMatch; i++) {
switch(editPath[i]) {
case 'delete':
// All deletions should appear at the sliding window edge; if a deletion appears
// after the edge, but before the first match, something's wrong.
if(priorEdit && priorEdit != 'delete') {
return failure;
}
leadTokensRemoved++;
break;
case 'substitute':
// Find the word before and after substitution.
const incomingIndex = i - (leadTokensRemoved > 0 ? leadTokensRemoved : 0);
const matchingIndex = i + (leadTokensRemoved < 0 ? leadTokensRemoved : 0);
const incomingSub = incomingTokenization[incomingIndex];
const matchingSub = tokenizationToMatch[matchingIndex];
const atSlidePoint = isSliding && (incomingIndex == 0 || matchingIndex == 0);
// Double-check the word - does the 'substituted' word itself align?
//
// Exception: if the word is at the start of the context window and the
// context window is likely sliding, don't check it.
if(!forAppliedSuggestion && !atSlidePoint && !isSubstitutionAlignable(incomingSub, matchingSub)) {
return failure;
}
leadSubstitutions++;
break;
case 'insert':
// Only allow an insert at the leading edge, as with 'delete's.
if(priorEdit && priorEdit != 'insert') {
return failure;
}
// In case of backspaces, it's also possible to 'insert' a 'new'
// token - an old one that's slid back into view.
leadTokensRemoved--;
break;
default:
// No 'match' can exist before the first found index for a 'match'.
// No 'transpose-' edits should exist within this section, either.
return failure;
}
priorEdit = editPath[i];
}
// If we need some form of tail-token substitution verification, add that here.
return {
canAlign: true,
editPath: editPaths[0],
// leadTokensRemoved represents the number of tokens that must be removed from the base context
// when aligning the contexts. Externally, it's more helpful to think in terms of the count added
// to the incoming context.
leadTokenShift: -leadTokensRemoved + 0, // add 0 in case of a 'negative zero', which affects unit tests.
leadEditLength: leadSubstitutions,
matchLength,
tailEditLength: tailSubstituteLength,
tailTokenShift: tailInsertLength - tailDeleteLength
};
}

View file

@ -9,39 +9,7 @@
*/
import { assert } from 'chai';
import { computeAlignment, EditOperation, getEditPathLastMatch, isSubstitutionAlignable } from '@keymanapp/lm-worker/test-index';
describe('getEditPathLastMatch', () => {
it('returns the last match when no substitutions exist', () => {
const path: EditOperation[] = ['delete', 'delete', 'match', 'match', 'match', 'match', 'insert'];
assert.equal(path.lastIndexOf('match'), 5);
assert.equal(getEditPathLastMatch(path), 5);
});
it('returns the last match when no substitutions exist left of a "match"', () => {
const path: EditOperation[] = ['delete', 'delete', 'match', 'match', 'match', 'match', 'substitute', 'insert'];
assert.equal(path.lastIndexOf('match'), 5);
assert.equal(getEditPathLastMatch(path), 5);
});
// is intended to handle application of suggestions.
it('returns the last match before a substitute occurring after the first match', () => {
// limitation: if there is _anything_ after that last match, the first assertion will fail.
// 0 1 2 3 4 5 6
const path: EditOperation[] = ['delete', 'delete', 'match', 'match', 'substitute', 'match', 'match'];
assert.notEqual(getEditPathLastMatch(path), 6);
assert.equal(getEditPathLastMatch(path), 3);
});
// is intended to handle complex transforms that include a whitespace and affect prior tokens.
it('returns the last match before a substitute occurring after the first match', () => {
// limitation: if there is _anything_ after that last match, the first assertion will fail.
// 0 1 2 3 4 5 6
const path: EditOperation[] = ['delete', 'delete', 'match', 'match', 'substitute', 'match', 'substitute'];
assert.notEqual(getEditPathLastMatch(path), 5);
assert.equal(getEditPathLastMatch(path), 3);
});
});
import { isSubstitutionAlignable } from '@keymanapp/lm-worker/test-index';
describe('isSubstitutionAlignable', () => {
it(`returns true: 'ca' => 'can'`, () => {
@ -120,773 +88,4 @@ describe('isSubstitutionAlignable', () => {
// The double-p adds a fun complication once the first gets dropped.
assert.isTrue(isSubstitutionAlignable('applesauce', 'plesauce'));
});
});
describe('computeAlignment', () => {
it("properly matches and aligns when contexts match", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [...baseContext];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'match', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'match', input: 4, match: 4}
],
leadTokenShift: 0,
leadEditLength: 0,
matchLength: 5,
tailEditLength: 0,
tailTokenShift: 0
});
});
it("properly matches and aligns with applied-suggestion contexts", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'o'
];
const newContext = [...baseContext];
newContext[4] = 'over';
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'match', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'substitute', input: 4, match: 4}
],
leadTokenShift: 0,
leadEditLength: 0,
matchLength: 4,
tailEditLength: 1,
tailTokenShift: 0
});
});
it("properly matches and aligns with applied-suggestion at start of context", () => {
const baseContext = [
'te'
];
const newContext = [
'testing',
' ',
''
];
const computedAlignment = computeAlignment(baseContext, newContext, false, true);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'substitute', input: 0, match: 0},
{op: 'insert', match: 1},
{op: 'insert', match: 2}
],
leadTokenShift: 0,
leadEditLength: 0,
matchLength: 0,
tailEditLength: 1,
tailTokenShift: 2
});
});
it("detects unalignable contexts - no matching tokens", () => {
const baseContext = [
'swift', 'tan', 'wolf', 'leaped', 'across'
];
const newContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: false,
editPath: [
{ op: 'substitute', input: 0, match: 0 },
{ op: 'substitute', input: 1, match: 1 },
{ op: 'substitute', input: 2, match: 2 },
{ op: 'substitute', input: 3, match: 3 },
{ op: 'substitute', input: 4, match: 4 }
]
});
});
it("detects unalignable contexts - too many mismatching tokens", () => {
const baseContext = [
'swift', 'tan', 'fox', 'jumped', 'over'
];
const newContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: false,
editPath: [
{op: 'substitute', input: 0, match: 0},
{op: 'substitute', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'match', input: 4, match: 4}
],
});
});
it("fails alignment for leading-edge word substitutions", () => {
const baseContext = [
'swift', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: false,
editPath: [
{op: 'substitute', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'match', input: 4, match: 4}
]
});
});
it("fails alignment for small leading-edge word substitutions", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'sick', 'brown', 'fox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: false,
editPath: [
{op: 'substitute', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'match', input: 4, match: 4}
]
});
});
it("properly matches and aligns when lead token is modified", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'uick', 'brown', 'fox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'substitute', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'match', input: 4, match: 4}
],
leadTokenShift: 0,
leadEditLength: 1,
matchLength: 4,
tailEditLength: 0,
tailTokenShift: 0
});
});
it("properly matches and aligns when lead token is removed", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'brown', 'fox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'delete', input: 0},
{op: 'match', input: 1, match: 0},
{op: 'match', input: 2, match: 1},
{op: 'match', input: 3, match: 2},
{op: 'match', input: 4, match: 3}
],
leadTokenShift: -1,
leadEditLength: 0,
matchLength: 4,
tailEditLength: 0,
tailTokenShift: 0
});
});
it("properly matches and aligns when lead token is added", () => {
const baseContext = [
'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'insert', match: 0},
{op: 'match', input: 0, match: 1},
{op: 'match', input: 1, match: 2},
{op: 'match', input: 2, match: 3},
{op: 'match', input: 3, match: 4}
],
leadTokenShift: 1,
leadEditLength: 0,
matchLength: 4,
tailEditLength: 0,
tailTokenShift: 0
});
});
it("properly matches and aligns when lead tokens are removed and modified", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'ox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'delete', input: 0},
{op: 'delete', input: 1},
{op: 'substitute', input: 2, match: 0},
{op: 'match', input: 3, match: 1},
{op: 'match', input: 4, match: 2},
],
leadTokenShift: -2,
leadEditLength: 1,
matchLength: 2,
tailEditLength: 0,
tailTokenShift: 0
});
});
it("properly matches and aligns when lead tokens are added and modified", () => {
const baseContext = [
'rown', 'fox', 'jumped', 'over'
];
const newContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'insert', match: 0},
{op: 'substitute', input: 0, match: 1},
{op: 'match', input: 1, match: 2},
{op: 'match', input: 2, match: 3},
{op: 'match', input: 3, match: 4},
],
leadTokenShift: 1,
leadEditLength: 1,
matchLength: 3,
tailEditLength: 0,
tailTokenShift: 0
});
});
it("properly matches and aligns when lead token is removed and tail token is added", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'brown', 'fox', 'jumped', 'over', 'the'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'delete', input: 0},
{op: 'match', input: 1, match: 0},
{op: 'match', input: 2, match: 1},
{op: 'match', input: 3, match: 2},
{op: 'match', input: 4, match: 3},
{op: 'insert', match: 4}
],
leadTokenShift: -1,
leadEditLength: 0,
matchLength: 4,
tailEditLength: 0,
tailTokenShift: 1
});
});
it("properly matches and aligns when lead token and tail token are modified", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'ove'
];
const newContext = [
'uick', 'brown', 'fox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'substitute', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'substitute', input: 4, match: 4}
],
leadTokenShift: 0,
leadEditLength: 1,
matchLength: 3,
tailEditLength: 1,
tailTokenShift: 0
});
});
it("properly matches and aligns when lead token and tail token are modified + new token appended", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'ove'
];
const newContext = [
'uick', 'brown', 'fox', 'jumped', 'over', 't'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'substitute', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'substitute', input: 4, match: 4},
{op: 'insert', match: 5}
],
leadTokenShift: 0,
leadEditLength: 1,
matchLength: 3,
tailEditLength: 1,
tailTokenShift: 1
});
});
it("properly handles context window sliding backward", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'e', 'quick', 'brown', 'fox', 'jumped', 'ove'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'insert', match: 0},
{op: 'match', input: 0, match: 1},
{op: 'match', input: 1, match: 2},
{op: 'match', input: 2, match: 3},
{op: 'match', input: 3, match: 4},
{op: 'substitute', input: 4, match: 5}
],
leadTokenShift: 1,
leadEditLength: 0,
matchLength: 4,
tailEditLength: 1,
tailTokenShift: 0
});
});
it("properly handles context window sliding far backward", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'the', 'quick', 'brown', 'fox', 'jumped'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'insert', match: 0},
{op: 'match', input: 0, match: 1},
{op: 'match', input: 1, match: 2},
{op: 'match', input: 2, match: 3},
{op: 'match', input: 3, match: 4},
{op: 'delete', input: 4}
],
leadTokenShift: 1,
leadEditLength: 0,
matchLength: 4,
tailEditLength: 0,
tailTokenShift: -1
});
});
it("properly handles context window sliding farther backward", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'the', 'quick', 'brown', 'fox', 'jumpe'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'insert', match: 0},
{op: 'match', input: 0, match: 1},
{op: 'match', input: 1, match: 2},
{op: 'match', input: 2, match: 3},
{op: 'substitute', input: 3, match: 4},
{op: 'delete', input: 4}
],
leadTokenShift: 1,
leadEditLength: 0,
matchLength: 3,
tailEditLength: 1,
tailTokenShift: -1
});
});
it("fails alignment for mid-head deletion", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'quick', 'fox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: false,
editPath: [
{op: 'match', input: 0, match: 0},
{op: 'delete', input: 1},
{op: 'match', input: 2, match: 1},
{op: 'match', input: 3, match: 2},
{op: 'match', input: 4, match: 3}
]
});
});
it("fails alignment for mid-head insertion", () => {
const baseContext = [
'quick', 'fox', 'jumped', 'over'
];
const newContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: false,
editPath: [
{op: 'match', input: 0, match: 0},
{op: 'insert', match: 1},
{op: 'match', input: 1, match: 2},
{op: 'match', input: 2, match: 3},
{op: 'match', input: 3, match: 4}
]
});
});
it("fails alignment for mid-tail deletion", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'quick', 'brown', 'fox', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: false,
editPath: [
{op: 'match', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'delete', input: 3},
{op: 'match', input: 4, match: 3}
]
});
});
it("fails alignment for mid-tail insertion", () => {
const baseContext = [
'quick', 'brown', 'fox', 'jumped', 'over'
];
const newContext = [
'quick', 'brown', 'fox', 'jumped', 'far', 'over'
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: false,
editPath: [
{op: 'match', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'insert', match: 4},
{op: 'match', input: 4, match: 5}
]
});
});
it("handles late-context suggestion application after backspace", () => {
const baseContext = [
'quick', ' ', 'brown', ' ', 'fox', ' ', 'jumped', ' ', 'oven', ' ', ''
];
const newContext = [
'quick', ' ', 'brown', ' ', 'fox', ' ', 'jumped', ' ', 'over', ' ', ''
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'match', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'match', input: 4, match: 4},
{op: 'match', input: 5, match: 5},
{op: 'match', input: 6, match: 6},
{op: 'match', input: 7, match: 7},
{op: 'substitute', input: 8, match: 8},
{op: 'match', input: 9, match: 9},
{op: 'match', input: 10, match: 10}
],
leadTokenShift: 0,
leadEditLength: 0,
matchLength: 8,
tailEditLength: 3,
tailTokenShift: 0
});
});
it("handles late-context application of default suggestion", () => {
const baseContext = [
'quick', ' ', 'brown', ' ', 'fox', ' ', 'jumped', ' ', 'over', ' ', ''
];
const newContext = [
'quick', ' ', 'brown', ' ', 'fox', ' ', 'jumped', ' ', 'over', ' ', 'the', ' ', ''
];
const computedAlignment = computeAlignment(baseContext, newContext, false);
assert.deepEqual(computedAlignment, {
canAlign: true,
editPath: [
{op: 'match', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'match', input: 4, match: 4},
{op: 'match', input: 5, match: 5},
{op: 'match', input: 6, match: 6},
{op: 'match', input: 7, match: 7},
{op: 'match', input: 8, match: 8},
{op: 'match', input: 9, match: 9},
{op: 'substitute', input: 10, match: 10},
{op: 'insert', match: 11},
{op: 'insert', match: 12}
],
leadTokenShift: 0,
leadEditLength: 0,
matchLength: 10,
tailEditLength: 1,
tailTokenShift: 2
});
});
it("handles sliding context-window scenarios", () => {
// // Explicitly-defined window, though it's not needed directly by the method.
// const config = {
// leftContextCodePoints: 64,
// rightContextCodePoints: 64
// };
const baseContext1 = [
// "ap" prefix not in actual view, but preserved by prior tokenization rounds.
"applesauce", " ", "and", " ", "orange", " ", "juice", " ", "don't", " ", "seem", " ",
"like", " ", "they'd", " ", "make", " ", "for", " ", "the", " ", "be"
];
const incomingContext1 = [
"plesauce", " ", "and", " ", "orange", " ", "juice", " ", "don't", " ", "seem", " ",
"like", " ", "they'd", " ", "make", " ", "for", " ", "the", " ", "bes"
];
// 66 chars above, vs a sliding window of length 64.
assert.equal(baseContext1.reduce((accum, curr) => accum + curr.length, 0), 66);
// Actual window + one newly-typed character
assert.equal(incomingContext1.reduce((accum, curr) => accum + curr.length, 0), 65);
assert.deepEqual(computeAlignment(baseContext1, incomingContext1, true), {
canAlign: true,
editPath: [
{op: 'substitute', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'match', input: 4, match: 4},
{op: 'match', input: 5, match: 5},
{op: 'match', input: 6, match: 6},
{op: 'match', input: 7, match: 7},
{op: 'match', input: 8, match: 8},
{op: 'match', input: 9, match: 9},
{op: 'match', input: 10, match: 10},
{op: 'match', input: 11, match: 11},
{op: 'match', input: 12, match: 12},
{op: 'match', input: 13, match: 13},
{op: 'match', input: 14, match: 14},
{op: 'match', input: 15, match: 15},
{op: 'match', input: 16, match: 16},
{op: 'match', input: 17, match: 17},
{op: 'match', input: 18, match: 18},
{op: 'match', input: 19, match: 19},
{op: 'match', input: 20, match: 20},
{op: 'match', input: 21, match: 21},
{op: 'substitute', input: 22, match: 22}
],
leadTokenShift: 0,
leadEditLength: 1,
matchLength: 21,
tailEditLength: 1,
tailTokenShift: 0
});
// Our tokenization scheme remembers the full original word before any of it slid out of
// the context window.
const baseContext2 = [
"applesauce", " ", "and", " ", "orange", " ", "juice", " ", "don't", " ", "seem", " ",
// +2 +1 +4
"like", " ", "they'd", " ", "make", " ", "for", " ", "the", " ", "best", " ", "brea"
];
const incomingContext2 = [
// "plesauce" => "e": -7 chars.
"e", " ", "and", " ", "orange", " ", "juice", " ", "don't", " ", "seem", " ",
"like", " ", "they'd", " ", "make", " ", "for", " ", "the", " ", "best", " ", "break"
];
// 73 chars above, vs a sliding window of length 64.
assert.equal(baseContext2.reduce((accum, curr) => accum + curr.length, 0), 73);
// Actual window + one newly-typed character
assert.equal(incomingContext2.reduce((accum, curr) => accum + curr.length, 0), 65);
assert.deepEqual(computeAlignment(baseContext2, incomingContext2, true), {
canAlign: true,
editPath: [
{op: 'substitute', input: 0, match: 0},
{op: 'match', input: 1, match: 1},
{op: 'match', input: 2, match: 2},
{op: 'match', input: 3, match: 3},
{op: 'match', input: 4, match: 4},
{op: 'match', input: 5, match: 5},
{op: 'match', input: 6, match: 6},
{op: 'match', input: 7, match: 7},
{op: 'match', input: 8, match: 8},
{op: 'match', input: 9, match: 9},
{op: 'match', input: 10, match: 10},
{op: 'match', input: 11, match: 11},
{op: 'match', input: 12, match: 12},
{op: 'match', input: 13, match: 13},
{op: 'match', input: 14, match: 14},
{op: 'match', input: 15, match: 15},
{op: 'match', input: 16, match: 16},
{op: 'match', input: 17, match: 17},
{op: 'match', input: 18, match: 18},
{op: 'match', input: 19, match: 19},
{op: 'match', input: 20, match: 20},
{op: 'match', input: 21, match: 21},
{op: 'match', input: 22, match: 22},
{op: 'match', input: 23, match: 23},
{op: 'substitute', input: 24, match: 24}
],
leadTokenShift: 0,
leadEditLength: 1,
matchLength: 23,
tailEditLength: 1,
tailTokenShift: 0
});
const baseContext3 = [
"applesauce", " ", "and", " ", "orange", " ", "juice", " ", "don't", " ", "seem", " ", "like", " ",
"they'd", " ", "make", " ", "for", " ", "the", " ", "best", " ", "break"
];
const incomingContext3 = [
" ", "and", " ", "orange", " ", "juice", " ", "don't", " ", "seem", " ", "like", " ",
"they'd", " ", "make", " ", "for", " ", "the", " ", "best", " ", "breakf"
];
// 74 chars above, vs a sliding window of length 64.
assert.equal(baseContext3.reduce((accum, curr) => accum + curr.length, 0), 74);
// Actual window + one newly-typed character
assert.equal(incomingContext3.reduce((accum, curr) => accum + curr.length, 0), 65);
assert.deepEqual(computeAlignment(baseContext3, incomingContext3, true), {
canAlign: true,
editPath: [
{op: 'delete', input: 0},
{op: 'match', input: 1, match: 0},
{op: 'match', input: 2, match: 1},
{op: 'match', input: 3, match: 2},
{op: 'match', input: 4, match: 3},
{op: 'match', input: 5, match: 4},
{op: 'match', input: 6, match: 5},
{op: 'match', input: 7, match: 6},
{op: 'match', input: 8, match: 7},
{op: 'match', input: 9, match: 8},
{op: 'match', input: 10, match: 9},
{op: 'match', input: 11, match: 10},
{op: 'match', input: 12, match: 11},
{op: 'match', input: 13, match: 12},
{op: 'match', input: 14, match: 13},
{op: 'match', input: 15, match: 14},
{op: 'match', input: 16, match: 15},
{op: 'match', input: 17, match: 16},
{op: 'match', input: 18, match: 17},
{op: 'match', input: 19, match: 18},
{op: 'match', input: 20, match: 19},
{op: 'match', input: 21, match: 20},
{op: 'match', input: 22, match: 21},
{op: 'match', input: 23, match: 22},
{op: 'substitute', input: 24, match: 23}
],
leadTokenShift: -1,
leadEditLength: 0,
matchLength: 23,
tailEditLength: 1,
tailTokenShift: 0
});
});
});