Merge pull request #1897 from keymanapp/lmlayer-specify-punctuation

[LMLayer] Allow model authors to specify quotes for keep message, and what to insert after a word
This commit is contained in:
Eddie Antonio Santos 2019-07-31 09:59:35 -06:00 • committed by GitHub
commit 41bf239f24
No known key found for this signature in database
GPG key ID: 4AEE18F83AFDEB23
14 changed files with 348 additions and 76 deletions

View file

@ -0,0 +1,104 @@
/*
* Unit tests for the Dummy prediction model.
*/
var assert = require('chai').assert;
var DummyModel = require('../../build/intermediate').models.DummyModel;
var ModelCompositor = require('../../build/intermediate').ModelCompositor;
describe('Custom Punctuation', function () {
it('appears in the keep suggestion', function () {
let dummySuggestions = [
{
transform: {
insert: 'Hrllo',
deleteLeft: 0,
},
tag: 'keep',
displayAs: 'Hrllo',
},
{
transform: {
insert: 'Hello',
deleteLeft: 0,
},
displayAs: 'Hello',
},
{
transform: {
insert: 'Jello',
deleteLeft: 0,
},
displayAs: 'Jello',
}
];
var model = new DummyModel({
futureSuggestions: [dummySuggestions],
punctuation: {
quotesForKeepSuggestion: {
open: "«", close: "»"
},
}
});
// The model compositor is responsible for adding this to the display as
// string.
var composite = new ModelCompositor(model);
var suggestions = composite.predict([{ sample: { insert: 'o', deleteLeft: 0 }, p: 1.00 }], {
left: 'Hrll', startOfBuffer: false, endOfBuffer: true
});
assert.lengthOf(suggestions, 3);
// We only care about the "keep" suggestion.
var keepSuggestions = suggestions.filter(suggestion => suggestion.tag === 'keep');
assert.lengthOf(keepSuggestions, 1,
`Expected exactly one "keep" suggestion, but found ${keepSuggestions.length}`
);
// The moment of truth: has our punctuation been applied?
var suggestion = keepSuggestions[0];
assert.equal(suggestion.displayAs, `«Hrllo»`);
});
describe("insertAfterWord", function () {
it('appears after "word" suggestion', function () {
let dummySuggestions = [
{
transform: { insert: 'ᚈᚑᚋ', deleteLeft: 0, },
displayAs: 'ᚈᚑᚋ',
},
{
transform: { insert: 'ᚄ', deleteLeft: 0, },
displayAs: 'ᚄ',
},
{
transform: { insert: 'ᚉᚑᚈᚈ', deleteLeft: 0, },
displayAs: 'ᚉᚑᚈᚈ',
}
];
var model = new DummyModel({
futureSuggestions: [dummySuggestions],
punctuation: {
// U+1680 OGHAM SPACE MARK:
// it's technically whitespace, but it don't look it!
insertAfterWord: " ",
}
});
// The model compositor is responsible for adding this to the display as
// string.
var composite = new ModelCompositor(model);
var suggestions = composite.predict([{ sample: { insert: 'ᚋ', deleteLeft: 0 }, p: 1.00 }], {
left: '᚛ᚈᚑ', startOfBuffer: false, endOfBuffer: true
});
assert.lengthOf(suggestions, dummySuggestions.length);
// Check that it has been changed:
for (var i = 0; i < dummySuggestions.length; i++) {
assert.isTrue(suggestions[i].transform.insert.endsWith(' '));
}
});
})
});

View file

@ -6,6 +6,27 @@ var assert = require('chai').assert;
var TrieModel = require('../../build/intermediate').models.TrieModel;
describe('LMLayerWorker trie model for word lists', function() {
describe('instantiation', function () {
it('should expose the punctuation object', function () {
var spaceMark = "👩🏻‍🚀";
var openQuote = "🌜";
var closeQuote = "🌛";
var model = new TrieModel(jsonFixture('tries/english-1000'), {
punctuation: {
insertAfterWord: spaceMark,
quotesForKeepSuggestion: {
open: openQuote, close: closeQuote
}
}
})
assert.equal(model.punctuation.insertAfterWord, spaceMark);
assert.equal(model.punctuation.quotesForKeepSuggestion.open, openQuote);
assert.equal(model.punctuation.quotesForKeepSuggestion.close, closeQuote);
})
});
describe('prediction', function () {
var MIN_SUGGESTIONS = 3;

View file

@ -19,7 +19,7 @@
"insert": "Oh ",
"deleteLeft": 0
},
"displayAs": "Oh "
"displayAs": "Oh"
}
],
[

View file

@ -7,6 +7,13 @@
function Model() { // implements Model
}
Model.punctuation = {
quotesForKeepSuggestion: { open: '“', close: '”'},
// Important! Set this, or else the model compositor will
// insert something for us!
insertAfterWord: "",
};
// A direct import/copy from i_got_distracted_by_hazel.json.
Model.futureSuggestions = [
[
@ -29,7 +36,7 @@
"insert": "Oh ",
"deleteLeft": 0
},
"displayAs": "Oh "
"displayAs": "Oh"
}
],
[
@ -107,5 +114,5 @@
}());
// It's a 'dummy' model, so there's no need for extra methods and such within the Model's class definition.
LMLayerWorker.loadModel(new models.DummyModel({futureSuggestions: Model.futureSuggestions}));
LMLayerWorker.loadModel(new models.DummyModel({futureSuggestions: Model.futureSuggestions, punctuation: Model.punctuation}));
})();

View file

@ -1,9 +1,11 @@
class ModelCompositor {
private lexicalModel: WorkerInternalModel;
private static readonly MAX_SUGGESTIONS = 12;
private readonly punctuation: LexicalModelPunctuation;
constructor(lexicalModel: WorkerInternalModel) {
this.lexicalModel = lexicalModel;
this.punctuation = ModelCompositor.determinePunctuationFromModel(lexicalModel);
}
protected isWhitespace(transform: Transform): boolean {
@ -28,6 +30,7 @@ class ModelCompositor {
predict(transformDistribution: Transform | Distribution<Transform>, context: Context): Suggestion[] {
let suggestionDistribution: Distribution<Suggestion> = [];
let punctuation = this.punctuation;
// Assumption: Duplicated 'displayAs' properties indicate duplicated Suggestions.
// When true, we can use an 'associative array' to de-duplicate everything.
@ -81,6 +84,12 @@ class ModelCompositor {
if(preserveWhitespace) {
models.prependTransform(pair.sample.transform, transform);
}
// The model is trying to add a word; thus, add some custom formatting
// to that word.
if (pair.sample.transform.insert.length > 0) {
pair.sample.transform.insert += punctuation.insertAfterWord;
}
// Combine duplicate samples.
let displayText = pair.sample.displayAs;
@ -109,7 +118,7 @@ class ModelCompositor {
transformId: inputTransform.id,
// Replicate the original transform, modified for appropriate language insertion syntax.
transform: {
insert: inputTransform.insert + ' ',
insert: inputTransform.insert + punctuation.insertAfterWord,
deleteLeft: inputTransform.deleteLeft,
deleteRight: inputTransform.deleteRight,
id: inputTransform.id
@ -118,10 +127,10 @@ class ModelCompositor {
};
}
// TODO: Customizable modeling for this formatting. Different languages
// use different quotation styles. See https://github.com/keymanapp/keyman/issues/1883.
if(keepOption) {
keepOption.displayAs = '"' + keepOption.displayAs + '"';
// Add the surrounding quotes to the "keep" option's display string:
if (keepOption) {
let { open, close } = punctuation.quotesForKeepSuggestion;
keepOption.displayAs = open + keepOption.displayAs + close;
}
// Now that we've calculated a unique set of probability masses, time to make them into a proper
@ -145,4 +154,38 @@ class ModelCompositor {
return suggestions;
}
}
/**
* Returns the punctuation used for this model, filling out unspecified fields
*/
private static determinePunctuationFromModel(model: WorkerInternalModel): LexicalModelPunctuation {
let defaults = DEFAULT_PUNCTUATION;
// Use the defaults of the model does not provide any punctuation at all.
if (!model.punctuation)
return defaults;
let specifiedPunctuation = model.punctuation;
let insertAfterWord = specifiedPunctuation.insertAfterWord;
if (insertAfterWord !== '' && !insertAfterWord) {
insertAfterWord = defaults.insertAfterWord;
}
let quotesForKeepSuggestion = specifiedPunctuation.quotesForKeepSuggestion;
if (!quotesForKeepSuggestion) {
quotesForKeepSuggestion = defaults.quotesForKeepSuggestion;
}
return {
insertAfterWord, quotesForKeepSuggestion
}
}
}
/**
* The default punctuation and spacing produced by the model.
*/
const DEFAULT_PUNCTUATION: LexicalModelPunctuation = {
quotesForKeepSuggestion: { open: `“`, close: `”`},
insertAfterWord: " " ,
};

View file

@ -33,6 +33,7 @@ namespace models {
*/
export class DummyModel implements WorkerInternalModel {
configuration: Configuration;
punctuation?: LexicalModelPunctuation;
private _futureSuggestions: Suggestion[][];
constructor(options?: any) {
@ -41,6 +42,10 @@ namespace models {
// this class mutates the array.
this._futureSuggestions = options.futureSuggestions
? options.futureSuggestions.slice() : [];
if (options.punctuation) {
this.punctuation = options.punctuation;
}
}
configure(capabilities: Capabilities): Configuration {

View file

@ -48,6 +48,11 @@
* This should simplify a search term into a key.
*/
searchTermToKey?: (searchTerm: string) => string;
/**
* Any punctuation to expose to the user.
*/
punctuation?: LexicalModelPunctuation;
}
/**
@ -72,6 +77,7 @@
configuration: Configuration;
private _trie: Trie;
readonly breakWords: WordBreakingFunction;
readonly punctuation?: LexicalModelPunctuation;
constructor(trieData: object, options: TrieModelOptions = {}) {
this._trie = new Trie(
@ -80,6 +86,7 @@
options.searchTermToKey as Wordform2Key || defaultWordform2Key
);
this.breakWords = options.wordBreaker || wordBreakers.placeholder;
this.punctuation = options.punctuation;
}
configure(capabilities: Capabilities): Configuration {
@ -94,7 +101,7 @@
if (!transform.insert && context.startOfBuffer && context.endOfBuffer) {
return makeDistribution(this._trie.firstN(MAX_SUGGESTIONS).map(({text, p}) => ({
transform: {
insert: text + ' ', // TODO: do NOT add the space here!
insert: text,
deleteLeft: 0
},
displayAs: text,

View file

@ -0,0 +1,80 @@
/**
* Types and interfaces that must be know both by the lexical model compiler,
* and by the runtime.
*/
/**
* A simple word breaking function takes a phrase, and splits it into "words",
* for whatever definition of "word" is usable for the language model.
*
* For example:
*
* getText(breakWordsEnglish("Hello, world!")) == ["Hello", "world"]
* getText(breakWordsCree("ᑕᐻ ᒥᔪ ᑮᓯᑲᐤ ᐊᓄᐦᐨ᙮")) == ["ᑕᐻ", "ᒥᔪ ᑮᓯᑲᐤ""", "ᐊᓄᐦᐨ"]
* getText(breakWordsJapanese("英語を話せますか?")) == ["英語", "を", "話せます", "か"]
*
* Not all language models take in a configurable word breaking function.
*
* @returns an array of spans from the phrase, in order as they appear in the
* phrase, each span which representing a word.
*/
interface WordBreakingFunction {
// invariant: span[i].end <= span[i + 1].start
// invariant: for all span[i] and span[i + 1], there does not exist a span[k]
// where span[i].end <= span[k].start AND span[k].end <= span[i + 1].start
(phrase: string): Span[];
}
/**
* A span of text in a phrase. This is usually meant to represent words from a
* pharse.
*/
interface Span {
// invariant: start < end (empty spans not allowed)
readonly start: number;
// invariant: end > end (empty spans not allowed)
readonly end: number;
// invariant: length === end - start
readonly length: number;
// invariant: text.length === length
// invariant: each character is BMP UTF-16 code unit, or is a high surrogate
// UTF-16 code unit followed by a low surrogate UTF-16 code unit.
readonly text: string;
}
/**
* Options for various punctuation to use in suggestions.
*/
interface LexicalModelPunctuation {
/**
* The quotes that appear in "keep" suggestions, e.g., keep what the user
* typed verbatim.
*
* The keep suggestion is often the leftmost one, when suggested.
*
* [ “Hrllo” ] [ Hello ] [ Heck ]
*/
readonly quotesForKeepSuggestion: {
/**
* What will appear on the opening side of the quote.
* (left side for LTR scripts; right side for RTL scripts)
*
* Default: `“`
*/
readonly open: string;
/**
* What will appear on the closing side of the quote.
* (right side for LTR scripts; left side for RTL scripts)
*
* Default: `”`
*/
readonly close: string;
};
/**
* What punctuation or spacing to insert after every complete word
* prediction. This can be set to the empty string when the script does not
* use spaces to separate words.
*
* Default: ` `
*/
readonly insertAfterWord: string;
}

View file

@ -27,6 +27,7 @@
*/
/// <reference path="../message.d.ts" />
/// <reference path="./worker-compiler-interfaces.d.ts" />
/**
* The signature of self.postMessage(), so that unit tests can mock it.
@ -184,6 +185,14 @@ interface WorkerInternalModel {
* @param context
*/
wordbreak(context: Context): USVString;
/**
* Punctuation and presentational settings that the underlying lexical model
* expects to be applied at higher levels. e.g., the ModelCompositor.
*
* @see LexicalModelPunctuation
*/
readonly punctuation?: LexicalModelPunctuation;
}
/**
@ -196,42 +205,3 @@ interface WorkerInternalModelConstructor {
*/
new(...modelParameters: any[]): WorkerInternalModel;
}
/**
* A simple word breaking function takes a phrase, and splits it into "words",
* for whatever definition of "word" is usable for the language model.
*
* For example:
*
* getText(breakWordsEnglish("Hello, world!")) == ["Hello", "world"]
* getText(breakWordsCree("ᑕᐻ ᒥᔪ ᑮᓯᑲᐤ ᐊᓄᐦᐨ᙮")) == ["ᑕᐻ", "ᒥᔪ ᑮᓯᑲᐤ""", "ᐊᓄᐦᐨ"]
* getText(breakWordsJapanese("英語を話せますか?")) == ["英語", "を", "話せます", "か"]
*
* Not all language models take in a configurable word breaking function.
*
* @returns an array of spans from the phrase, in order as they appear in the
* phrase, each span which representing a word.
*/
interface WordBreakingFunction {
// invariant: span[i].end <= span[i + 1].start
// invariant: for all span[i] and span[i + 1], there does not exist a span[k]
// where span[i].end <= span[k].start AND span[k].end <= span[i + 1].start
(phrase: USVString): Span[];
}
/**
* A span of text in a phrase. This is usually meant to reprent words from a
* pharse.
*/
interface Span {
// invariant: start < end (empty spans not allowed)
readonly start: number;
// invariant: end > end (empty spans not allowed)
readonly end: number;
// invariant: length === end - start
readonly length: number;
// invariant: text.length === length
// invariant: each character is BMP UTF-16 code unit, or is a high surrogate
// UTF-16 code unit followed by a low surrogate UTF-16 code unit.
readonly text: string;
}

View file

@ -231,11 +231,9 @@ export default class LexicalModelCompiler {
func += `LMLayerWorker.loadModel(new ${modelSource.rootClass}());\n`;
break;
case "fst-foma-1.0":
(oc as LexicalModelCompiledFst).fst = Buffer.from(sources.join('')).toString('base64');
this.logError('Unimplemented model format '+modelSource.format);
this.logError('Unimplemented model format ' + modelSource.format);
return false;
case "trie-1.0": // TODO: in lexical-models, rollback all trie-2.0 models to use trie-1.0!
case 'trie-2.0':
case "trie-1.0":
func += `LMLayerWorker.loadModel(new models.TrieModel(${
createTrieDataStructure(sources, modelSource.searchTermToKey)
}, {\n`;
@ -245,6 +243,9 @@ export default class LexicalModelCompiler {
if (modelSource.searchTermToKey) {
func += ` searchTermToKey: ${modelSource.searchTermToKey.toString()},\n`;
}
if (modelSource.punctuation) {
func += ` punctuation: ${JSON.stringify(modelSource.punctuation)},\n`;
}
func += `}));\n`;
break;
default:

View file

@ -1,3 +1,9 @@
/**
* Interfaces and constants used by the lexical model compiler. These target
* the LMLayer's internal worker code, so we provide those definitions too.
*/
/// <reference path="../../../common/predictive-text/worker/worker-compiler-interfaces.d.ts" />
interface ClassBasedWordBreaker {
allowedCharacters?: { initials?: string, medials?: string, finals?: string } | string,
defaultBreakCharacter?: string
@ -9,17 +15,10 @@ interface ClassBasedWordBreaker {
}
interface LexicalModel {
// TODO: remove trie-2.0 when https://github.com/keymanapp/lexical-models/pull/39 is merged.
readonly format: 'trie-1.0'|'trie-2.0'|'fst-foma-1.0'|'custom-1.0',
readonly format: 'trie-1.0'|'fst-foma-1.0'|'custom-1.0',
//... metadata ...
}
interface LexicalModelPrediction {
display?: string;
transform: string;
delete: number;
}
interface LexicalModelSource extends LexicalModel {
readonly sources: Array<string>;
/**
@ -35,19 +34,15 @@ interface LexicalModelSource extends LexicalModel {
* This often involves removing accents, lowercasing, etc.
*/
readonly searchTermToKey?: (term: string) => string;
/**
* Punctuation and spacing suggested by the model.
*
* @see LexicalModelPunctuation
*/
readonly punctuation?: LexicalModelPunctuation;
}
interface LexicalModelCompiled extends LexicalModel {
readonly id: string;
}
interface LexicalModelCompiledTrie extends LexicalModelCompiled {
trie: string;
}
interface LexicalModelCompiledFst extends LexicalModelCompiled {
fst: string;
}
interface LexicalModelCompiledCustom extends LexicalModelCompiled {
}

View file

@ -0,0 +1,32 @@
import LexicalModelCompiler from '../';
import {assert} from 'chai';
import 'mocha';
import path = require('path');
describe('LexicalModelCompiler', function () {
describe('spefifying punctuation', function () {
const MODEL_ID = 'example.qaa.trivial';
const PATH = path.join(__dirname, 'fixtures', MODEL_ID)
it('should compile punctuation into the generated code', function () {
let compiler = new LexicalModelCompiler;
let code = compiler.generateLexicalModelCode(MODEL_ID, {
format: 'trie-1.0',
sources: ['wordlist.tsv'],
punctuation: {
quotesForKeepSuggestion: { open: `«`, close: `»`},
insertAfterWord: " " , // OGHAM SPACE MARK
}
}, PATH) as string;
// Check that the punctuation actually made into the code:
assert.match(code, /«/);
assert.match(code, /»/);
// Ensure we inserted that OGHAM SPACE MARK!
assert.match(code, /\u1680/);
// TODO: more robust assertions?
});
})
});

View file

@ -0,0 +1,10 @@
{
"compilerOptions": {
"module": "commonjs",
"noImplicitAny": true,
"sourceMap": true
},
"include": [
"**/test-*.ts"
]
}

View file

@ -1,13 +1,10 @@
{
"compilerOptions": {
"target": "es6",
"target": "es2017",
"module": "commonjs",
"outDir": "dist",
"sourceMap": true,
"declaration": true,
"typeRoots": [
"./node_modules/@types"
]
},
"include": [
"lexical-model-compiler/**/*",