diff --git a/common/predictive-text/unit_tests/headless/worker-custom-punctuation.js b/common/predictive-text/unit_tests/headless/worker-custom-punctuation.js new file mode 100644 index 0000000000..3407ace41f --- /dev/null +++ b/common/predictive-text/unit_tests/headless/worker-custom-punctuation.js @@ -0,0 +1,104 @@ +/* + * Unit tests for the Dummy prediction model. + */ + +var assert = require('chai').assert; +var DummyModel = require('../../build/intermediate').models.DummyModel; +var ModelCompositor = require('../../build/intermediate').ModelCompositor; + +describe('Custom Punctuation', function () { + it('appears in the keep suggestion', function () { + let dummySuggestions = [ + { + transform: { + insert: 'Hrllo', + deleteLeft: 0, + }, + tag: 'keep', + displayAs: 'Hrllo', + }, + { + transform: { + insert: 'Hello', + deleteLeft: 0, + }, + displayAs: 'Hello', + }, + { + transform: { + insert: 'Jello', + deleteLeft: 0, + }, + displayAs: 'Jello', + } + ]; + + var model = new DummyModel({ + futureSuggestions: [dummySuggestions], + punctuation: { + quotesForKeepSuggestion: { + open: "«", close: "»" + }, + } + }); + + // The model compositor is responsible for adding this to the display as + // string. + var composite = new ModelCompositor(model); + var suggestions = composite.predict([{ sample: { insert: 'o', deleteLeft: 0 }, p: 1.00 }], { + left: 'Hrll', startOfBuffer: false, endOfBuffer: true + }); + assert.lengthOf(suggestions, 3); + + // We only care about the "keep" suggestion. + var keepSuggestions = suggestions.filter(suggestion => suggestion.tag === 'keep'); + assert.lengthOf(keepSuggestions, 1, + `Expected exactly one "keep" suggestion, but found ${keepSuggestions.length}` + ); + + // The moment of truth: has our punctuation been applied? + var suggestion = keepSuggestions[0]; + assert.equal(suggestion.displayAs, `«Hrllo»`); + }); + + describe("insertAfterWord", function () { + it('appears after "word" suggestion', function () { + let dummySuggestions = [ + { + transform: { insert: 'ᚈᚑᚋ', deleteLeft: 0, }, + displayAs: 'ᚈᚑᚋ', + }, + { + transform: { insert: 'ᚄ', deleteLeft: 0, }, + displayAs: 'ᚄ', + }, + { + transform: { insert: 'ᚉᚑᚈᚈ', deleteLeft: 0, }, + displayAs: 'ᚉᚑᚈᚈ', + } + ]; + + var model = new DummyModel({ + futureSuggestions: [dummySuggestions], + punctuation: { + // U+1680 OGHAM SPACE MARK: + // it's technically whitespace, but it don't look it! + insertAfterWord: " ", + } + }); + + // The model compositor is responsible for adding this to the display as + // string. + var composite = new ModelCompositor(model); + var suggestions = composite.predict([{ sample: { insert: 'ᚋ', deleteLeft: 0 }, p: 1.00 }], { + left: '᚛ᚈᚑ', startOfBuffer: false, endOfBuffer: true + }); + assert.lengthOf(suggestions, dummySuggestions.length); + + // Check that it has been changed: + for (var i = 0; i < dummySuggestions.length; i++) { + assert.isTrue(suggestions[i].transform.insert.endsWith(' ')); + } + }); + }) +}); \ No newline at end of file diff --git a/common/predictive-text/unit_tests/headless/worker-predict-trie.js b/common/predictive-text/unit_tests/headless/worker-predict-trie.js index a19994f803..cde71da62c 100644 --- a/common/predictive-text/unit_tests/headless/worker-predict-trie.js +++ b/common/predictive-text/unit_tests/headless/worker-predict-trie.js @@ -6,6 +6,27 @@ var assert = require('chai').assert; var TrieModel = require('../../build/intermediate').models.TrieModel; describe('LMLayerWorker trie model for word lists', function() { + describe('instantiation', function () { + it('should expose the punctuation object', function () { + var spaceMark = "👩🏻‍🚀"; + var openQuote = "🌜"; + var closeQuote = "🌛"; + + var model = new TrieModel(jsonFixture('tries/english-1000'), { + punctuation: { + insertAfterWord: spaceMark, + quotesForKeepSuggestion: { + open: openQuote, close: closeQuote + } + } + }) + + assert.equal(model.punctuation.insertAfterWord, spaceMark); + assert.equal(model.punctuation.quotesForKeepSuggestion.open, openQuote); + assert.equal(model.punctuation.quotesForKeepSuggestion.close, closeQuote); + }) + }); + describe('prediction', function () { var MIN_SUGGESTIONS = 3; diff --git a/common/predictive-text/unit_tests/in_browser/json/future_suggestions/i_got_distracted_by_hazel.json b/common/predictive-text/unit_tests/in_browser/json/future_suggestions/i_got_distracted_by_hazel.json index 347b98d2e6..a9aa49753b 100644 --- a/common/predictive-text/unit_tests/in_browser/json/future_suggestions/i_got_distracted_by_hazel.json +++ b/common/predictive-text/unit_tests/in_browser/json/future_suggestions/i_got_distracted_by_hazel.json @@ -19,7 +19,7 @@ "insert": "Oh ", "deleteLeft": 0 }, - "displayAs": "Oh " + "displayAs": "Oh" } ], [ diff --git a/common/predictive-text/unit_tests/in_browser/resources/models/simple-dummy.js b/common/predictive-text/unit_tests/in_browser/resources/models/simple-dummy.js index 85d7b3057d..97180729bd 100644 --- a/common/predictive-text/unit_tests/in_browser/resources/models/simple-dummy.js +++ b/common/predictive-text/unit_tests/in_browser/resources/models/simple-dummy.js @@ -7,6 +7,13 @@ function Model() { // implements Model } + Model.punctuation = { + quotesForKeepSuggestion: { open: '“', close: '”'}, + // Important! Set this, or else the model compositor will + // insert something for us! + insertAfterWord: "", + }; + // A direct import/copy from i_got_distracted_by_hazel.json. Model.futureSuggestions = [ [ @@ -29,7 +36,7 @@ "insert": "Oh ", "deleteLeft": 0 }, - "displayAs": "Oh " + "displayAs": "Oh" } ], [ @@ -107,5 +114,5 @@ }()); // It's a 'dummy' model, so there's no need for extra methods and such within the Model's class definition. - LMLayerWorker.loadModel(new models.DummyModel({futureSuggestions: Model.futureSuggestions})); + LMLayerWorker.loadModel(new models.DummyModel({futureSuggestions: Model.futureSuggestions, punctuation: Model.punctuation})); })(); \ No newline at end of file diff --git a/common/predictive-text/worker/model-compositor.ts b/common/predictive-text/worker/model-compositor.ts index 00dfc66997..8331336e9c 100644 --- a/common/predictive-text/worker/model-compositor.ts +++ b/common/predictive-text/worker/model-compositor.ts @@ -1,9 +1,11 @@ class ModelCompositor { private lexicalModel: WorkerInternalModel; private static readonly MAX_SUGGESTIONS = 12; + private readonly punctuation: LexicalModelPunctuation; constructor(lexicalModel: WorkerInternalModel) { this.lexicalModel = lexicalModel; + this.punctuation = ModelCompositor.determinePunctuationFromModel(lexicalModel); } protected isWhitespace(transform: Transform): boolean { @@ -28,6 +30,7 @@ class ModelCompositor { predict(transformDistribution: Transform | Distribution, context: Context): Suggestion[] { let suggestionDistribution: Distribution = []; + let punctuation = this.punctuation; // Assumption: Duplicated 'displayAs' properties indicate duplicated Suggestions. // When true, we can use an 'associative array' to de-duplicate everything. @@ -81,6 +84,12 @@ class ModelCompositor { if(preserveWhitespace) { models.prependTransform(pair.sample.transform, transform); } + + // The model is trying to add a word; thus, add some custom formatting + // to that word. + if (pair.sample.transform.insert.length > 0) { + pair.sample.transform.insert += punctuation.insertAfterWord; + } // Combine duplicate samples. let displayText = pair.sample.displayAs; @@ -109,7 +118,7 @@ class ModelCompositor { transformId: inputTransform.id, // Replicate the original transform, modified for appropriate language insertion syntax. transform: { - insert: inputTransform.insert + ' ', + insert: inputTransform.insert + punctuation.insertAfterWord, deleteLeft: inputTransform.deleteLeft, deleteRight: inputTransform.deleteRight, id: inputTransform.id @@ -118,10 +127,10 @@ class ModelCompositor { }; } - // TODO: Customizable modeling for this formatting. Different languages - // use different quotation styles. See https://github.com/keymanapp/keyman/issues/1883. - if(keepOption) { - keepOption.displayAs = '"' + keepOption.displayAs + '"'; + // Add the surrounding quotes to the "keep" option's display string: + if (keepOption) { + let { open, close } = punctuation.quotesForKeepSuggestion; + keepOption.displayAs = open + keepOption.displayAs + close; } // Now that we've calculated a unique set of probability masses, time to make them into a proper @@ -145,4 +154,38 @@ class ModelCompositor { return suggestions; } -} \ No newline at end of file + + /** + * Returns the punctuation used for this model, filling out unspecified fields + */ + private static determinePunctuationFromModel(model: WorkerInternalModel): LexicalModelPunctuation { + let defaults = DEFAULT_PUNCTUATION; + + // Use the defaults of the model does not provide any punctuation at all. + if (!model.punctuation) + return defaults; + + let specifiedPunctuation = model.punctuation; + let insertAfterWord = specifiedPunctuation.insertAfterWord; + if (insertAfterWord !== '' && !insertAfterWord) { + insertAfterWord = defaults.insertAfterWord; + } + + let quotesForKeepSuggestion = specifiedPunctuation.quotesForKeepSuggestion; + if (!quotesForKeepSuggestion) { + quotesForKeepSuggestion = defaults.quotesForKeepSuggestion; + } + + return { + insertAfterWord, quotesForKeepSuggestion + } + } +} + +/** + * The default punctuation and spacing produced by the model. + */ +const DEFAULT_PUNCTUATION: LexicalModelPunctuation = { + quotesForKeepSuggestion: { open: `“`, close: `”`}, + insertAfterWord: " " , +}; diff --git a/common/predictive-text/worker/models/dummy-model.ts b/common/predictive-text/worker/models/dummy-model.ts index 4052fc5482..06809b3fa5 100644 --- a/common/predictive-text/worker/models/dummy-model.ts +++ b/common/predictive-text/worker/models/dummy-model.ts @@ -33,6 +33,7 @@ namespace models { */ export class DummyModel implements WorkerInternalModel { configuration: Configuration; + punctuation?: LexicalModelPunctuation; private _futureSuggestions: Suggestion[][]; constructor(options?: any) { @@ -41,6 +42,10 @@ namespace models { // this class mutates the array. this._futureSuggestions = options.futureSuggestions ? options.futureSuggestions.slice() : []; + + if (options.punctuation) { + this.punctuation = options.punctuation; + } } configure(capabilities: Capabilities): Configuration { diff --git a/common/predictive-text/worker/models/trie-model.ts b/common/predictive-text/worker/models/trie-model.ts index 97b4df91b0..57ca0ead14 100644 --- a/common/predictive-text/worker/models/trie-model.ts +++ b/common/predictive-text/worker/models/trie-model.ts @@ -48,6 +48,11 @@ * This should simplify a search term into a key. */ searchTermToKey?: (searchTerm: string) => string; + + /** + * Any punctuation to expose to the user. + */ + punctuation?: LexicalModelPunctuation; } /** @@ -72,6 +77,7 @@ configuration: Configuration; private _trie: Trie; readonly breakWords: WordBreakingFunction; + readonly punctuation?: LexicalModelPunctuation; constructor(trieData: object, options: TrieModelOptions = {}) { this._trie = new Trie( @@ -80,6 +86,7 @@ options.searchTermToKey as Wordform2Key || defaultWordform2Key ); this.breakWords = options.wordBreaker || wordBreakers.placeholder; + this.punctuation = options.punctuation; } configure(capabilities: Capabilities): Configuration { @@ -94,7 +101,7 @@ if (!transform.insert && context.startOfBuffer && context.endOfBuffer) { return makeDistribution(this._trie.firstN(MAX_SUGGESTIONS).map(({text, p}) => ({ transform: { - insert: text + ' ', // TODO: do NOT add the space here! + insert: text, deleteLeft: 0 }, displayAs: text, diff --git a/common/predictive-text/worker/worker-compiler-interfaces.d.ts b/common/predictive-text/worker/worker-compiler-interfaces.d.ts new file mode 100644 index 0000000000..05d432c95a --- /dev/null +++ b/common/predictive-text/worker/worker-compiler-interfaces.d.ts @@ -0,0 +1,80 @@ +/** + * Types and interfaces that must be know both by the lexical model compiler, + * and by the runtime. + */ + +/** + * A simple word breaking function takes a phrase, and splits it into "words", + * for whatever definition of "word" is usable for the language model. + * + * For example: + * + * getText(breakWordsEnglish("Hello, world!")) == ["Hello", "world"] + * getText(breakWordsCree("ᑕᐻ ᒥᔪ ᑮᓯᑲᐤ ᐊᓄᐦᐨ᙮")) == ["ᑕᐻ", "ᒥᔪ ᑮᓯᑲᐤ""", "ᐊᓄᐦᐨ"] + * getText(breakWordsJapanese("英語を話せますか?")) == ["英語", "を", "話せます", "か"] + * + * Not all language models take in a configurable word breaking function. + * + * @returns an array of spans from the phrase, in order as they appear in the + * phrase, each span which representing a word. + */ +interface WordBreakingFunction { + // invariant: span[i].end <= span[i + 1].start + // invariant: for all span[i] and span[i + 1], there does not exist a span[k] + // where span[i].end <= span[k].start AND span[k].end <= span[i + 1].start + (phrase: string): Span[]; +} + +/** + * A span of text in a phrase. This is usually meant to represent words from a + * pharse. + */ +interface Span { + // invariant: start < end (empty spans not allowed) + readonly start: number; + // invariant: end > end (empty spans not allowed) + readonly end: number; + // invariant: length === end - start + readonly length: number; + // invariant: text.length === length + // invariant: each character is BMP UTF-16 code unit, or is a high surrogate + // UTF-16 code unit followed by a low surrogate UTF-16 code unit. + readonly text: string; +} +/** + * Options for various punctuation to use in suggestions. + */ +interface LexicalModelPunctuation { + /** + * The quotes that appear in "keep" suggestions, e.g., keep what the user + * typed verbatim. + * + * The keep suggestion is often the leftmost one, when suggested. + * + * [ “Hrllo” ] [ Hello ] [ Heck ] + */ + readonly quotesForKeepSuggestion: { + /** + * What will appear on the opening side of the quote. + * (left side for LTR scripts; right side for RTL scripts) + * + * Default: `“` + */ + readonly open: string; + /** + * What will appear on the closing side of the quote. + * (right side for LTR scripts; left side for RTL scripts) + * + * Default: `”` + */ + readonly close: string; + }; + /** + * What punctuation or spacing to insert after every complete word + * prediction. This can be set to the empty string when the script does not + * use spaces to separate words. + * + * Default: ` ` + */ + readonly insertAfterWord: string; +} diff --git a/common/predictive-text/worker/worker-interfaces.ts b/common/predictive-text/worker/worker-interfaces.ts index 69f4d7bcef..eabef62841 100644 --- a/common/predictive-text/worker/worker-interfaces.ts +++ b/common/predictive-text/worker/worker-interfaces.ts @@ -27,6 +27,7 @@ */ /// +/// /** * The signature of self.postMessage(), so that unit tests can mock it. @@ -184,6 +185,14 @@ interface WorkerInternalModel { * @param context */ wordbreak(context: Context): USVString; + + /** + * Punctuation and presentational settings that the underlying lexical model + * expects to be applied at higher levels. e.g., the ModelCompositor. + * + * @see LexicalModelPunctuation + */ + readonly punctuation?: LexicalModelPunctuation; } /** @@ -196,42 +205,3 @@ interface WorkerInternalModelConstructor { */ new(...modelParameters: any[]): WorkerInternalModel; } - -/** - * A simple word breaking function takes a phrase, and splits it into "words", - * for whatever definition of "word" is usable for the language model. - * - * For example: - * - * getText(breakWordsEnglish("Hello, world!")) == ["Hello", "world"] - * getText(breakWordsCree("ᑕᐻ ᒥᔪ ᑮᓯᑲᐤ ᐊᓄᐦᐨ᙮")) == ["ᑕᐻ", "ᒥᔪ ᑮᓯᑲᐤ""", "ᐊᓄᐦᐨ"] - * getText(breakWordsJapanese("英語を話せますか?")) == ["英語", "を", "話せます", "か"] - * - * Not all language models take in a configurable word breaking function. - * - * @returns an array of spans from the phrase, in order as they appear in the - * phrase, each span which representing a word. - */ -interface WordBreakingFunction { - // invariant: span[i].end <= span[i + 1].start - // invariant: for all span[i] and span[i + 1], there does not exist a span[k] - // where span[i].end <= span[k].start AND span[k].end <= span[i + 1].start - (phrase: USVString): Span[]; -} - -/** - * A span of text in a phrase. This is usually meant to reprent words from a - * pharse. - */ -interface Span { - // invariant: start < end (empty spans not allowed) - readonly start: number; - // invariant: end > end (empty spans not allowed) - readonly end: number; - // invariant: length === end - start - readonly length: number; - // invariant: text.length === length - // invariant: each character is BMP UTF-16 code unit, or is a high surrogate - // UTF-16 code unit followed by a low surrogate UTF-16 code unit. - readonly text: string; -} diff --git a/developer/js/index.ts b/developer/js/index.ts index e53ffbe838..5ef9ea5c5f 100644 --- a/developer/js/index.ts +++ b/developer/js/index.ts @@ -231,11 +231,9 @@ export default class LexicalModelCompiler { func += `LMLayerWorker.loadModel(new ${modelSource.rootClass}());\n`; break; case "fst-foma-1.0": - (oc as LexicalModelCompiledFst).fst = Buffer.from(sources.join('')).toString('base64'); - this.logError('Unimplemented model format '+modelSource.format); + this.logError('Unimplemented model format ' + modelSource.format); return false; - case "trie-1.0": // TODO: in lexical-models, rollback all trie-2.0 models to use trie-1.0! - case 'trie-2.0': + case "trie-1.0": func += `LMLayerWorker.loadModel(new models.TrieModel(${ createTrieDataStructure(sources, modelSource.searchTermToKey) }, {\n`; @@ -245,6 +243,9 @@ export default class LexicalModelCompiler { if (modelSource.searchTermToKey) { func += ` searchTermToKey: ${modelSource.searchTermToKey.toString()},\n`; } + if (modelSource.punctuation) { + func += ` punctuation: ${JSON.stringify(modelSource.punctuation)},\n`; + } func += `}));\n`; break; default: diff --git a/developer/js/lexical-model-compiler/lexical-model.ts b/developer/js/lexical-model-compiler/lexical-model.ts index 115cdaf792..92aa4c7261 100644 --- a/developer/js/lexical-model-compiler/lexical-model.ts +++ b/developer/js/lexical-model-compiler/lexical-model.ts @@ -1,3 +1,9 @@ +/** + * Interfaces and constants used by the lexical model compiler. These target + * the LMLayer's internal worker code, so we provide those definitions too. + */ +/// + interface ClassBasedWordBreaker { allowedCharacters?: { initials?: string, medials?: string, finals?: string } | string, defaultBreakCharacter?: string @@ -9,17 +15,10 @@ interface ClassBasedWordBreaker { } interface LexicalModel { - // TODO: remove trie-2.0 when https://github.com/keymanapp/lexical-models/pull/39 is merged. - readonly format: 'trie-1.0'|'trie-2.0'|'fst-foma-1.0'|'custom-1.0', + readonly format: 'trie-1.0'|'fst-foma-1.0'|'custom-1.0', //... metadata ... } -interface LexicalModelPrediction { - display?: string; - transform: string; - delete: number; -} - interface LexicalModelSource extends LexicalModel { readonly sources: Array; /** @@ -35,19 +34,15 @@ interface LexicalModelSource extends LexicalModel { * This often involves removing accents, lowercasing, etc. */ readonly searchTermToKey?: (term: string) => string; + + /** + * Punctuation and spacing suggested by the model. + * + * @see LexicalModelPunctuation + */ + readonly punctuation?: LexicalModelPunctuation; } interface LexicalModelCompiled extends LexicalModel { readonly id: string; } - -interface LexicalModelCompiledTrie extends LexicalModelCompiled { - trie: string; -} - -interface LexicalModelCompiledFst extends LexicalModelCompiled { - fst: string; -} - -interface LexicalModelCompiledCustom extends LexicalModelCompiled { -} diff --git a/developer/js/tests/test-punctuation.ts b/developer/js/tests/test-punctuation.ts new file mode 100644 index 0000000000..e8f24bdba3 --- /dev/null +++ b/developer/js/tests/test-punctuation.ts @@ -0,0 +1,32 @@ +import LexicalModelCompiler from '../'; +import {assert} from 'chai'; +import 'mocha'; + +import path = require('path'); + + +describe('LexicalModelCompiler', function () { + describe('spefifying punctuation', function () { + const MODEL_ID = 'example.qaa.trivial'; + const PATH = path.join(__dirname, 'fixtures', MODEL_ID) + + it('should compile punctuation into the generated code', function () { + let compiler = new LexicalModelCompiler; + let code = compiler.generateLexicalModelCode(MODEL_ID, { + format: 'trie-1.0', + sources: ['wordlist.tsv'], + punctuation: { + quotesForKeepSuggestion: { open: `«`, close: `»`}, + insertAfterWord: " " , // OGHAM SPACE MARK + } + }, PATH) as string; + + // Check that the punctuation actually made into the code: + assert.match(code, /«/); + assert.match(code, /»/); + // Ensure we inserted that OGHAM SPACE MARK! + assert.match(code, /\u1680/); + // TODO: more robust assertions? + }); + }) +}); diff --git a/developer/js/tests/tsconfig.json b/developer/js/tests/tsconfig.json new file mode 100644 index 0000000000..655a3fcfe1 --- /dev/null +++ b/developer/js/tests/tsconfig.json @@ -0,0 +1,10 @@ +{ + "compilerOptions": { + "module": "commonjs", + "noImplicitAny": true, + "sourceMap": true + }, + "include": [ + "**/test-*.ts" + ] +} \ No newline at end of file diff --git a/developer/js/tsconfig.json b/developer/js/tsconfig.json index 901da41ac0..0a05d8ae77 100644 --- a/developer/js/tsconfig.json +++ b/developer/js/tsconfig.json @@ -1,13 +1,10 @@ { "compilerOptions": { - "target": "es6", + "target": "es2017", "module": "commonjs", "outDir": "dist", "sourceMap": true, "declaration": true, - "typeRoots": [ - "./node_modules/@types" - ] }, "include": [ "lexical-model-compiler/**/*",