Merge pull request #1766 from keymanapp/developer-move-lexical-model-compiler-to-keyman-repo

[Developer] Move lexical model compiler to Keyman repo
This commit is contained in:
Marc Durdin 2019-05-08 13:42:55 +10:00 committed by GitHub
commit bd1dcb70a4
No known key found for this signature in database
GPG key ID: 4AEE18F83AFDEB23
13 changed files with 1243 additions and 0 deletions

27
developer/README.md Normal file
View file

@ -0,0 +1,27 @@
# Keyman Developer
This is the likely future home for Keyman Developer. However, at this
time, Keyman Developer still lives in windows/src/developer/
This folder currently contains only the Lexical Model Compiler and the
related Package Compiler. It runs on nodeJS on all supported desktop
platforms.
# Lexical Model Compiler
# Package Compiler
The package compiler is broadly compatible with the kmcomp .kps
package compiler. However at this stage it is only tested with
lexical models, and use with keyboards (either .js or .kmx) is not
tested or supported. It is likely in the future that the kmcomp
.kps compiler will be deprecated in favour of this one.
# Transition steps
1. Move both of the above compilers into the keyman repo
2. Split .kps and model compilers into separate paths
3. Update lexical-models repo to pull these compiler(s)
4. Refactor compiler code to address https://github.com/keymanapp/lexical-models/issues/28 and https://github.com/keymanapp/lexical-models/issues/29

2
developer/js/.gitignore vendored Normal file
View file

@ -0,0 +1,2 @@
node_modules/
dist/

245
developer/js/index.ts Normal file
View file

@ -0,0 +1,245 @@
/*
index.ts: base file for lexical model compiler.
*/
/// <reference path="./lexical-model-compiler/lexical-model.ts" />
/// <reference path="./lexical-model-compiler/model-info-file.ts" />
import * as ts from "typescript";
import KmpCompiler from "./package-compiler/kmp-compiler";
import * as fs from "fs";
import * as path from "path";
import { createWordListDataStructure, createTrieDataStructure } from "./lexical-model-compiler/build-trie";
// The model ID MUST adhere to this pattern:
// author .bcp47 .uniq
const MODEL_ID_PATTERN = /^[a-z_][a-z0-9_]*\.[a-z_][a-z0-9_]*\.[a-z_][a-z0-9_]*$/;
export default class LexicalModelCompiler {
compile(modelSource: LexicalModelSource) {
//
// Load the model info file
//
let files = fs.readdirSync('../');
let model_info_file = files.find((f) => !!f.match(/\.model_info$/));
if(!model_info_file) {
this.logError('Unable to find .model_info file in parent folder');
return false;
}
let model_id = model_info_file.match(/^(.+)\.model_info$/)[1];
if(!model_id.match(MODEL_ID_PATTERN)) {
this.logError(
`The model identifier '${model_id}' is invalid.\n`+
`Must be a valid alphanumeric identifier in format (author).(bcp_47).(uniq).\n`+
`bcp_47 should be underscore (_) separated.`);
return false;
}
/*
* Model info looks like this:
*
* {
* "name": "Example Template Model"
* "license": "mit",
* "version": "1.0.0",
* "languages": ["en"],
* "authorName": "Example Author",
* "authorEmail": "nobody@example.com",
* "description": "Example wordlist model"
* }
*
* For full documentation, see:
* https://help.keyman.com/developer/cloud/model_info/1.0/
*/
let model_info: ModelInfoFile = JSON.parse(fs.readFileSync('../'+model_info_file, 'utf8'));
//
// Filename expectations
//
const kpsFileName = `../source/${model_id}.model.kps`;
const kmpFileName = `${model_id}.model.kmp`;
const modelFileName = `${model_id}.model.js`;
const modelInfoFileName = `${model_id}.model_info`;
const sourcePath = '../source';
const minKeymanVersion = '12.0';
//
// Validate the model ID.
//
//
// This script is run from folder group/author/bcp47.uniq/build/ folder. We want to
// verify that author.bcp47.uniq is the same as the model identifier.
//
let paths = process.cwd().split(path.sep).reverse();
if(paths.length < 4 || paths[0] != 'build' || model_id != paths[2] + '.' + paths[1]) {
this.logError(`Unexpected model path ${paths[2]}.${paths[1]}, does not match model id ${model_id}`);
return false;
}
// 0 = build
// 1 = bcp47.uniq
// 2 = author
// 3 = group
let groupPath = paths[3];
let authorPath = paths[2];
let bcp47Path = paths[1];
//
// Build the compiled lexical model
//
let sources: string[] = modelSource.sources.map(function(source) {
return fs.readFileSync(path.join(sourcePath, source), 'utf8');
});
let oc: LexicalModelCompiled = {id: model_id, format: modelSource.format};
// TODO: add metadata in comment
const filePrefix: string = `(function() {\n'use strict';\n`;
const fileSuffix: string = `})();`;
let func = filePrefix;
let wordBreakingSource: string = null;
if (modelSource.wordBreaking) {
if (typeof modelSource.wordBreaking === "string") {
// It must be a builtin word breaker, so just instantiate it.
wordBreakingSource = `wordBreakers['${modelSource.wordBreaking}']`;
} else if (modelSource.wordBreaking.sources) {
let wordBreakingSources: string[] = modelSource.wordBreaking.sources.map(function(source) {
return fs.readFileSync(path.join(sourcePath, source), 'utf8');
});
wordBreakingSource = this.transpileSources(wordBreakingSources).join('\n');
}
}
//
// Emit the model as code and data
//
switch(modelSource.format) {
case "custom-1.0":
func += this.transpileSources(sources).join('\n');
// JSON.stringify(oc) gives the base metadata
func += `LMLayerWorker.loadModel(new ${modelSource.rootClass}());\n`;
break;
case "fst-foma-1.0":
(oc as LexicalModelCompiledFst).fst = Buffer.from(sources.join('')).toString('base64');
this.logError('Unimplemented model format '+modelSource.format);
return false;
case "trie-1.0":
func += `var model = {};\n`;
func += `model.backingData = ${createWordListDataStructure(sources)};\n`;
func += `LMLayerWorker.loadModel(new models.WordListModel(model.backingData`;
if (wordBreakingSource) {
func += `, {wordBreaking: ${wordBreakingSource}}`;
}
func += `));\n`;
break;
case 'trie-2.0':
// TODO: allow specification of key function.
func += `LMLayerWorker.loadModel(new models.TrieModel(${
createTrieDataStructure(sources)
}`;
if (wordBreakingSource) {
func += `, {wordBreaking: ${wordBreakingSource}}`;
}
func += `));\n`;
break;
default:
this.logError('Unknown model format '+modelSource.format);
return false;
}
//
// Load custom wordbreak source files
//
func += fileSuffix;
// Save full model to build folder as Javascript for use in KeymanWeb
fs.writeFileSync(modelFileName, func);
//
// Create KMP file
//
let kpsString: string = fs.readFileSync(kpsFileName, 'utf8');
let kmpCompiler = new KmpCompiler();
let kmpJsonData = kmpCompiler.transformKpsToKmpObject(model_id, kpsString);
kmpCompiler.buildKmpFile(kmpJsonData, kmpFileName);
//
// Build merged .model_info file
// https://api.keyman.com/schemas/model_info.source.json and
// https://api.keyman.com/schemas/model_info.distribution.json
// https://help.keyman.com/developer/cloud/model_info/1.0
//
function set_model_metadata(field: string, expected: any, warn: boolean = true) {
if(model_info[field] && model_info[field] !== expected) {
if(warn || typeof warn === 'undefined')
console.warn(`Warning: source ${modelInfoFileName} field ${field} value "${model_info[field]}" does not match "${expected}" found in source file metadata.`);
}
model_info[field] = model_info[field] || expected;
}
// Merge model info file -- some fields have "special" behaviours -- see below
set_model_metadata('id', model_id);
set_model_metadata('name', kmpJsonData.info.name.description);
set_model_metadata('authorName', kmpJsonData.info.author.description);
// we strip the mailto: from the .kps file for the .model_info
set_model_metadata('authorEmail', kmpJsonData.info.author.url.match(/^(mailto\:)?(.+)$/)[2], false);
// extract the language identifiers from the language metadata
// arrays for each of the lexical models in the kmp.json file,
// and merge into a single array of identifiers in the
// .model_info file.
model_info.languages = model_info.languages || [].concat(kmpJsonData.lexicalModels.map((e) => e.languages.map((f) => f.id)));
set_model_metadata('lastModifiedDate', (new Date).toISOString());
set_model_metadata('packageFilename', kmpFileName);
// Always overwrite with actual file size
model_info.packageFileSize = fs.statSync(model_info.packageFilename).size;
set_model_metadata('jsFilename', modelFileName);
// Always overwrite with actual file size
model_info.jsFileSize = fs.statSync(model_info.jsFilename).size;
// Always overwrite source data
model_info.packageIncludes = kmpJsonData.files.filter((e) => !!e.name.match(/.[ot]tf$/i)).length ? ['fonts'] : [];
set_model_metadata('version', kmpJsonData.info.version.description);
// The minimum Keyman version detected in the package file may be manually set higher by the developer
set_model_metadata('minKeymanVersion', minKeymanVersion, false);
//TODO: model_info.helpLink = model_info.helpLink || ... if source/help/id.php exists?
set_model_metadata('sourcePath', [groupPath, authorPath, bcp47Path].join('/'));
fs.writeFileSync(modelInfoFileName, JSON.stringify(model_info, null, 2));
};
transpileSources(sources: Array<string>): Array<string> {
return sources.map((source) => ts.transpileModule(source, {
compilerOptions: { module: ts.ModuleKind.None }
}).outputText
);
};
logError(s) {
console.error(require('chalk').red(s));
};
};

View file

@ -0,0 +1,371 @@
/**
* A word list is an array of pairs: the concrete word form itself, followed by
* a non-negative count.
*/
type WordList = [string, number][];
/**
* Returns a data structure suitable for use by the wordlist model.
*
* @param sourceFiles an array of the CONTENTS of source files
*
* @return a data structure that will be used internally by the wordlist
* implemention. Currently this is an array of [wordlist, count] pairs.
*/
export function createWordListDataStructure(sourceFiles: string[]): string {
// NOTE: this generates a simple array of word forms --- not a trie!
// In the future, this function may construct a true trie data structure,
// but this is not yet implemented.
let contents = sourceFiles.join('\n');
return JSON.stringify(parseWordList(contents));
}
/**
* Returns a data structure that can be loaded by the TrieModel.
*
* It implements a **weighted** trie, whose indices (paths down the trie) are
* generated by a search key, and not concrete wordforms themselves.
*
* @param sourceFiles an array of source files that will be read to generate the trie.
*/
export function createTrieDataStructure(sourceFiles: string[]): string {
let wordlist = parseWordList(sourceFiles.join('\n'));
let trie = Trie.buildTrie(wordlist);
return JSON.stringify(trie);
}
/**
* Reads a tab-separated values file into a word list.
*
* Format specification:
*
* - the file is a UTF-8 encoded text file
* - new lines are either LF or CRLF
* - the file either consists of a comment or an entry
* - comment lines MUST start with the '#' character on the very first column
* - entries are one to three columns, separated by the (horizontal) tab
* character
* - column 1 (REQUIRED): the wordform: can have any character except tab, CR,
* LF. Surrounding whitespace characters are trimmed.
* - column 2 (optional): the count: a non-negative integer specifying how many
* times this entry has appeared in the corpus. Blank means 'indeterminate'
* - column 3 (optional): comment: an informative comment, ignored by the tool.
*/
export function parseWordList(contents: string): WordList {
// Supports LF or CRLF line terminators.
const NEWLINE_SEPARATOR = /\u000d?\u000a/;
const TAB = "\t";
// TODO: format validation.
let lines = contents.split(NEWLINE_SEPARATOR);
let result: WordList = [];
for (let line of lines) {
if (line.startsWith('#') || line === "") {
continue; // skip comments and empty lines
}
let [wordform, countText, _comment] = line.split(TAB);
// Clean the word form.
// TODO: what happens if we get duplicate forms?
wordform = wordform.normalize('NFC').trim();
countText = (countText || '').trim();
let count = parseInt(countText, 10);
// When parsing a decimal integer fails (e.g., blank or something else):
if (!isFinite(count)) {
// TODO: is this the right thing to do?
// Treat it like a hapax legonmenom -- it exist, but only once.
count = 1;
}
result.push([wordform, count]);
}
return result;
}
namespace Trie {
/**
* An **opaque** type for a string that is exclusively used as a search key in
* the trie. There should be a function that converts arbitrary strings
* (queries) and converts them into a standard search key for a given language
* model.
*
* Fun fact: This opaque type has ALREADY saved my bacon and found a bug!
*/
type SearchKey = string & { _: 'SearchKey'};
/**
* A function that converts a string (word form or query) into a search key
* (secretly, this is also a string).
*/
interface Wordform2Key {
(wordform: string): SearchKey;
}
// The following trie implementation has been (heavily) derived from trie-ing
// by Conrad Irwin.
//
// trie-ing is distributed under the terms of the MIT license, reproduced here:
//
// The MIT License
// Copyright (c) 2015-2017 Conrad Irwin <conrad.irwin@gmail.com>
// Copyright (c) 2011 Marc Campbell <marc.e.campbell@gmail.com>
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
//
// See: https://github.com/ConradIrwin/trie-ing/blob/df55d7af7068d357829db9e0a7faa8a38add1d1d/LICENSE
/**
* An entry in the prefix trie. The matched word is "content".
*/
interface Entry {
content: string;
key: SearchKey;
weight: number;
}
/**
* The trie is made up of nodes. A node can be EITHER an internal node (whose
* only children are other nodes) OR a leaf, which actually contains the word
* form entries.
*/
type Node = InternalNode | Leaf;
/**
* An internal node.
*/
interface InternalNode {
type: 'internal';
weight: number;
// TODO: As an optimization, "values" can be a single string!
values: string[];
children: { [codeunit: string]: Node };
unsorted?: true;
}
/**
* A leaf node.
*/
interface Leaf {
type: 'leaf';
weight: number;
entries: Entry[];
unsorted?: true;
}
/**
* A sentinel value for when an internal node has contents and requires an
* "internal" leaf. That is, this internal node has content. Instead of placing
* entries as children in an internal node, a "fake" leaf is created, and its
* key is this special internal value.
*
* The value is a valid Unicode BMP code point, but it is a "non-character".
* Unicode will never assign semantics to these characters, as they are
* intended to be used internally as sentinel values.
*/
const INTERNAL_VALUE = '\uFDD0';
/**
* Builds a trie from a word list.
*
* @param wordlist The wordlist with non-negative weights.
* @param keyFunction Function that converts word forms into indexed search keys
* @returns The root node as a JSON-serialiable object.
*/
export function buildTrie(wordlist: WordList, keyFunction: Wordform2Key = defaultWordform2Key): object {
return new Trie(keyFunction).buildFromWordList(wordlist).root;
}
/**
* Wrapper class for the trie and its nodes and wordform to search
*/
class Trie {
readonly root = createRootNode();
toKey: Wordform2Key;
constructor(wordform2key: Wordform2Key) {
this.toKey = wordform2key;
}
/**
* Populates the trie with the contents of an entire wordlist.
* @param words a list of word and count pairs.
*/
buildFromWordList(words: WordList): Trie {
for (let [wordform, weight] of words) {
let key = this.toKey(wordform);
addUnsorted(this.root, { key, weight, content: wordform }, 0);
}
sortTrie(this.root);
return this;
}
}
// "Constructors"
function createRootNode(): Node {
return {
type: 'leaf',
weight: 0,
entries: []
};
}
// Implement Trie creation.
/**
* Adds an entry to the trie.
*
* Note that the trie will likely be unsorted after the add occurs. Before
* performing a lookup on the trie, use call sortTrie() on the root note!
*
* @param node Which node should the entry be added to?
* @param entry the wordform/weight/key to add to the trie
* @param index the index in the key and also the trie depth. Should be set to
* zero when adding onto the root node of the trie.
*/
function addUnsorted(node: Node, entry: Entry, index: number = 0) {
// Each node stores the MAXIMUM weight out of all of its decesdents, to
// enable a greedy search through the trie.
node.weight = Math.max(node.weight, entry.weight);
// When should a leaf become an interior node?
// When it already has a value, but the key of the current value is longer
// than the prefix.
if (node.type === 'leaf' && index < entry.key.length && node.entries.length >= 1) {
convertLeafToInternalNode(node, index);
}
if (node.type === 'leaf') {
// The key matches this leaf node, so add yet another entry.
addItemToLeaf(node, entry);
} else {
// Push the node down to a lower node.
addItemToInternalNode(node, entry, index);
}
node.unsorted = true;
}
/**
* Adds an item to the internal node at a given depth.
* @param item
* @param index
*/
function addItemToInternalNode(node: InternalNode, item: Entry, index: number) {
let char = item.key[index];
if (!node.children[char]) {
node.children[char] = createRootNode();
node.values.push(char);
}
addUnsorted(node.children[char], item, index + 1);
}
function addItemToLeaf(leaf: Leaf, item: Entry) {
leaf.entries.push(item);
}
/**
* Mutates the given Leaf to turn it into an InternalNode.
*
* NOTE: the node passed in will be DESTRUCTIVELY CHANGED into a different
* type when passed into this function!
*
* @param depth depth of the trie at this level.
*/
function convertLeafToInternalNode(leaf: Leaf, depth: number): void {
let entries = leaf.entries;
// Alias the current node, as the desired type.
let internal = (<unknown> leaf) as InternalNode;
internal.type = 'internal';
delete leaf.entries;
internal.values = [];
internal.children = {};
// Convert the old values array into the format for interior nodes.
for (let item of entries) {
let char: string;
if (depth < item.key.length) {
char = item.key[depth];
} else {
char = INTERNAL_VALUE;
}
if (!internal.children[char]) {
internal.children[char] = createRootNode();
internal.values.push(char);
}
addUnsorted(internal.children[char], item, depth + 1);
}
internal.unsorted = true;
}
/**
* Recursively sort the trie, in descending order of weight.
* @param node any node in the trie
*/
function sortTrie(node: Node) {
if (node.type === 'leaf') {
if (!node.unsorted) {
return;
}
node.entries.sort(function (a, b) { return b.weight - a.weight; });
} else {
// We MUST recurse and sort children before returning.
for (let char of node.values) {
sortTrie(node.children[char]);
}
if (!node.unsorted) {
return;
}
node.values.sort((a, b) => {
return node.children[b].weight - node.children[a].weight;
});
}
delete node.unsorted;
}
/**
* Converts word forms in into an indexable form. It does this by converting
* the string to uppercase and trying to remove diacritical marks.
*
* This is a very naïve implementation, that I've only though to work on
* languages that use the Latin script. Even then, some Latin-based
* orthographies use code points that, under NFD normalization, do NOT
* decompose into an ASCII letter and a combining diacritical mark (e.g.,
* SENĆOŦEN).
*
* Use this only in early iterations of the model. For a production lexical
* model, you SHOULD write/generate your own key function, tailored to your
* language.
*/
function defaultWordform2Key(wordform: string): SearchKey {
// Use this pattern to remove common diacritical marks.
// See: https://www.compart.com/en/unicode/block/U+0300
const COMBINING_DIACRITICAL_MARKS = /[\u0300-\u036f]/g;
return wordform
.normalize('NFD')
.toUpperCase()
// remove diacritical marks.
.replace(COMBINING_DIACRITICAL_MARKS, '') as SearchKey;
}
}

View file

@ -0,0 +1,44 @@
interface ClassBasedWordBreaker {
allowedCharacters?: { initials?: string, medials?: string, finals?: string } | string,
defaultBreakCharacter?: string
sources?: Array<string>;
/**
* The name of the type to instantiate (without parameters) as the base object for a custom word-breaking model.
*/
rootClass?: string
}
interface LexicalModel {
readonly format: 'trie-1.0'|'trie-2.0'|'fst-foma-1.0'|'custom-1.0',
//... metadata ...
}
interface LexicalModelPrediction {
display?: string;
transform: string;
delete: number;
}
interface LexicalModelSource extends LexicalModel {
readonly sources: Array<string>;
/**
* The name of the type to instantiate (without parameters) as the base object for a custom predictive model.
*/
readonly rootClass?: string
readonly wordBreaking?: 'ascii' | 'placeholder' | ClassBasedWordBreaker;
}
interface LexicalModelCompiled extends LexicalModel {
readonly id: string;
}
interface LexicalModelCompiledTrie extends LexicalModelCompiled {
trie: string;
}
interface LexicalModelCompiledFst extends LexicalModelCompiled {
fst: string;
}
interface LexicalModelCompiledCustom extends LexicalModelCompiled {
}

View file

@ -0,0 +1,33 @@
interface ModelInfoFile {
id?: string;
name?: string;
authorName?: string;
authorEmail?: string;
description?: string;
license: "mit";
languages: Array<string>;
lastModifiedDate?: string;
links: ModelInfoFileLink[];
packageFilename?: string;
packageFileSize?: number;
jsFilename?: string;
jsFileSize?: number;
isRTL?: boolean;
packageIncludes?: string[]; //['fonts'] or []
version?: string;
minKeymanVersion?: string;
helpLink?: string;
sourcePath?: string;
related?: ModelInfoFileRelated[];
}
interface ModelInfoFileLink {
name: string;
url: string;
}
interface ModelInfoFileRelated {
deprecates?: string;
deprecatedBy?: string;
note?: string;
}

View file

@ -0,0 +1,7 @@
// This file defines stub callbacks so that dev-time work on wordbreaker and predictors don't give
// errors.
// TODO: Turn into a .d.ts?
// TODO: jsdoc function comments and parameter names
let com = {keyman: {lexicalModel: { registerWordBreaker(a,b) {}, registerPredictor(a,b) {} } } };

View file

@ -0,0 +1,141 @@
// <reference path="lexical-model.ts" />
/// <reference path="kps-file.ts" />
/// <reference path="kmp-json-file.ts" />
let fs = require('fs');
let path = require('path');
let xml2js = require('xml2js');
let zip = require('node-zip')();
export default class KmpCompiler {
public transformKpsToKmpObject(modelId: string, kpsString: string): KmpJsonFile {
// Load the KPS data from XML as JS structured data.
let kpsPackage = (() => {
let a;
let parser = new xml2js.Parser({
tagNameProcessors: [xml2js.processors.firstCharLowerCase],
explicitArray: false
});
parser.parseString(kpsString, (e, r) => { a = r });
return a;
})();
let kps: KpsFile = kpsPackage.package as KpsFile;
//
// To convert to kmp.json, we need to:
//
// 1. Unwrap arrays (and convert to array where single object)
// 2. Fix casing on `iD`
// 3. Rewrap info, keyboard.languages, lexicalModel.languages, startMenu.items elements
// 4. Convert options.followKeyboardVersion to a bool
// 5. Filenames need to be basenames (but this comes after processing)
//
// Helper functions
let kpsInfoToKmpInfo = function (info: KpsFileInfo): KmpJsonFileInfo {
let ni: KmpJsonFileInfo = {};
['author', 'copyright', 'name', 'version', 'website'].forEach(element => {
if(info[element]) {
ni[element] = {description: info[element]._ || info[element]};
if(info[element].$ && info[element].$.URL) ni[element].url = info[element].$.URL;
}
});
return ni;
};
let arrayWrap = function(a) {
if(Array.isArray(a)) {
return a;
}
return [a];
};
let kpsLanguagesToKmpLanguages = function(language: KpsFileLanguage[]): KmpJsonFileLanguage[] {
return language.map((element) => { return { name: element._, id: element.$.ID } });
};
// Start to construct the kmp.json file from the .kps file
let kmp: KmpJsonFile = {
system: kps.system,
options: {
followKeyboardVersion: kps.options.followKeyboardVersion === ''
}
};
// Fill in additional fields
['executeProgram', 'graphicFile', 'msiFilename', 'msiOptions', 'readmeFile'].forEach((element) => {
if(kps.options[element]) kmp.options[element] = kps.options[element];
});
if(kps.info) {
kmp.info = kpsInfoToKmpInfo(kps.info);
}
if(kps.files && kps.files.file) {
kmp.files = arrayWrap(kps.files.file);
}
if(kps.keyboards && kps.keyboards.keyboard) {
kmp.keyboards = arrayWrap(kps.keyboards.keyboard).map((keyboard: KpsFileKeyboard) => {
return { name:keyboard.name, id:keyboard.iD, version:keyboard.version, languages: kpsLanguagesToKmpLanguages(arrayWrap(keyboard.languages.language) as KpsFileLanguage[]) }
});
}
if(kps.lexicalModels && kps.lexicalModels.lexicalModel) {
kmp.lexicalModels = arrayWrap(kps.lexicalModels.lexicalModel).map((model: KpsFileLexicalModel) => {
return { name:model.name, id:model.iD, version:model.version, languages: kpsLanguagesToKmpLanguages(arrayWrap(model.languages.language) as KpsFileLanguage[]) }
});
}
if(kps.startMenu) {
kmp.startMenu = {};
if(kps.startMenu.addUninstallEntry) kmp.startMenu.addUninstallEntry = kps.startMenu.addUninstallEntry === '';
if(kps.startMenu.folder) kmp.startMenu.folder = kps.startMenu.folder;
if(kps.startMenu.items && kps.startMenu.items.item) kmp.startMenu.items = arrayWrap(kps.startMenu.items.item);
}
if(kps.strings && kps.strings.string) {
kmp.strings = arrayWrap(kps.strings.string);
}
//let util = require('util');
//console.log(util.inspect(kmp, false, null))
//console.log(kps);
return kmp;
}
public buildKmpFile(kmpJsonData: KmpJsonFile, kmpFileName: string) {
const kmpJsonFileName = 'kmp.json';
if(!kmpJsonData.files) {
kmpJsonData.files = [];
}
kmpJsonData.files.forEach(function(value) {
// Make file path slashes compatible across platforms
let filename : string = value.name.replace(/\\/g, "/");
let data = fs.readFileSync(path.join('../source', filename), 'utf8');
zip.file(path.basename(filename), data);
// Remove path data from files before save
value.name = path.basename(filename);
});
zip.file(kmpJsonFileName, JSON.stringify(kmpJsonData));
// Generate kmp file
var data = zip.generate({base64:false,compression:'DEFLATE'});
fs.writeFileSync(kmpFileName, data, 'binary');
}
}

View file

@ -0,0 +1,79 @@
interface KmpJsonFile {
system: KmpJsonFileSystem;
options: KmpJsonFileOptions;
info?: KmpJsonFileInfo;
files?: KmpJsonFileContentFile[];
lexicalModels?: KmpJsonFileLexicalModel[];
startMenu?: KmpJsonFileStartMenu;
keyboards?: KmpJsonFileKeyboard[];
strings?: string[];
}
interface KmpJsonFileSystem {
keymanDeveloperVersion: string;
fileVersion: string;
}
interface KmpJsonFileOptions {
followKeyboardVersion: boolean;
readmeFile?: string;
graphicFile?: string;
executeProgram?: string;
msiFilename?: string;
msiOptions?: string;
}
interface KmpJsonFileInfo {
website?: KmpJsonFileInfoItem;
version?: KmpJsonFileInfoItem;
name?: KmpJsonFileInfoItem;
copyright?: KmpJsonFileInfoItem;
author?: KmpJsonFileInfoItem;
}
interface KmpJsonFileInfoItem {
description: string;
url?: string;
}
interface KmpJsonFileContentFile {
name: string;
description: string;
copyLocation?: number;
}
interface KmpJsonFileLexicalModel {
name: string;
id: string;
version: string;
languages: KmpJsonFileLanguage[];
}
interface KmpJsonFileLanguage {
name: string;
id: string;
}
interface KmpJsonFileKeyboard {
name: string;
id: string;
version: string;
oskFont?: string;
displayFont?: string;
rtl?: boolean;
languages?: KmpJsonFileLanguage[];
}
interface KmpJsonFileStartMenu {
folder?: string;
addUninstallEntry?: boolean;
items?: KmpJsonFileStartMenuItem[];
}
interface KmpJsonFileStartMenuItem {
name: string;
filename: string;
arguments?: string;
icon?: string;
location?: string;
}

View file

@ -0,0 +1,120 @@
//
// The interfaces in this file are designed with reference to the
// mapped structures produced by xml2js when passed a .kps file.
//
// A few notes:
//
// * Casing is updated to camelCase during load (leaving `iD` as a
// mixed up beastie).
// * Arrays are buried a layer too deep (e.g. <Files><File/><File/></Files>
// leads to KpsFiles.KpsFile[]
// * Properties such as used in Info Items use `_` and `$` and must be
// extracted.
// * Strings element is not yet checked to be correct
//
interface KpsFile {
system: KpsFileSystem;
options: KpsFileOptions;
info?: KpsFileInfo;
files?: KpsFileContentFiles;
keyboards?: KpsFileKeyboards;
lexicalModels?: KpsFileLexicalModels;
startMenu?: KpsFileStartMenu;
strings?: KpsFileStrings;
}
interface KpsFileSystem {
keymanDeveloperVersion: string;
fileVersion: string;
}
interface KpsFileOptions {
followKeyboardVersion?: string;
readmeFile?: string;
graphicFile?: string;
executeProgram?: string;
msiFilename?: string;
msiOptions?: string;
}
interface KpsFileInfo {
name?: KpsFileInfoItem;
copyright?: KpsFileInfoItem;
author?: KpsFileInfoItem;
website?: KpsFileInfoItem;
version?: KpsFileInfoItem;
}
interface KpsFileInfoItem {
_: string;
$: { URL: string };
}
interface KpsFileContentFiles {
file: KpsFileContentFile[] | KpsFileContentFile;
}
interface KpsFileContentFile {
name: string;
description: string;
copyLocation: string;
fileType: string;
}
interface KpsFileLexicalModel {
name: string;
iD: string;
version: string;
languages: KpsFileLanguages;
}
interface KpsFileLexicalModels {
lexicalModel: KpsFileLexicalModel[] | KpsFileLexicalModel;
}
interface KpsFileLanguages {
language: KpsFileLanguage[] | KpsFileLanguage;
}
interface KpsFileLanguage {
_: string;
$: { ID: string }
}
interface KpsFileKeyboard {
name: string;
iD: string;
version: string;
oskFont?: string;
displayFont?: string;
rtl?: boolean;
languages?: KpsFileLanguages;
}
interface KpsFileKeyboards {
keyboard: KpsFileKeyboard[] | KpsFileKeyboard;
}
interface KpsFileStartMenu {
folder?: string;
addUninstallEntry?: string;
items?: KpsFileStartMenuItems;
}
interface KpsFileStartMenuItem {
name: string;
filename: string;
arguments?: string;
icon?: string;
location?: string;
}
interface KpsFileStartMenuItems {
item: KpsFileStartMenuItem[] | KpsFileStartMenuItem;
}
interface KpsFileStrings {
//TODO: validate this structure
string: string[] | string;
}

131
developer/js/package-lock.json generated Normal file
View file

@ -0,0 +1,131 @@
{
"name": "@keymanapp/developer-lexical-model-compiler",
"version": "1.0.0",
"lockfileVersion": 1,
"requires": true,
"dependencies": {
"@types/node": {
"version": "10.14.6",
"resolved": "https://registry.npmjs.org/@types/node/-/node-10.14.6.tgz",
"integrity": "sha512-Fvm24+u85lGmV4hT5G++aht2C5I4Z4dYlWZIh62FAfFO/TfzXtPpoLI6I7AuBWkIFqZCnhFOoTT7RjjaIL5Fjg==",
"dev": true
},
"ansi-styles": {
"version": "3.2.1",
"resolved": "https://registry.npmjs.org/ansi-styles/-/ansi-styles-3.2.1.tgz",
"integrity": "sha512-VT0ZI6kZRdTh8YyJw3SMbYm/u+NqfsAxEpWO0Pf9sq8/e94WxxOpPKx9FR1FlyCtOVDNOQ+8ntlqFxiRc+r5qA==",
"dev": true,
"requires": {
"color-convert": "^1.9.0"
}
},
"chalk": {
"version": "2.4.2",
"resolved": "https://registry.npmjs.org/chalk/-/chalk-2.4.2.tgz",
"integrity": "sha512-Mti+f9lpJNcwF4tWV8/OrTTtF1gZi+f8FqlyAdouralcFWFQWF2+NgCHShjkCb+IFBLq9buZwE1xckQU4peSuQ==",
"dev": true,
"requires": {
"ansi-styles": "^3.2.1",
"escape-string-regexp": "^1.0.5",
"supports-color": "^5.3.0"
}
},
"color-convert": {
"version": "1.9.3",
"resolved": "https://registry.npmjs.org/color-convert/-/color-convert-1.9.3.tgz",
"integrity": "sha512-QfAUtd+vFdAtFQcC8CCyYt1fYWxSqAiK2cSD6zDB8N3cpsEBAvRxp9zOGg6G/SHHJYAT88/az/IuDGALsNVbGg==",
"dev": true,
"requires": {
"color-name": "1.1.3"
}
},
"color-name": {
"version": "1.1.3",
"resolved": "https://registry.npmjs.org/color-name/-/color-name-1.1.3.tgz",
"integrity": "sha1-p9BVi9icQveV3UIyj3QIMcpTvCU=",
"dev": true
},
"escape-string-regexp": {
"version": "1.0.5",
"resolved": "https://registry.npmjs.org/escape-string-regexp/-/escape-string-regexp-1.0.5.tgz",
"integrity": "sha1-G2HAViGQqN/2rjuyzwIAyhMLhtQ=",
"dev": true
},
"has-flag": {
"version": "3.0.0",
"resolved": "https://registry.npmjs.org/has-flag/-/has-flag-3.0.0.tgz",
"integrity": "sha1-tdRU3CGZriJWmfNGfloH87lVuv0=",
"dev": true
},
"jszip": {
"version": "2.5.0",
"resolved": "https://registry.npmjs.org/jszip/-/jszip-2.5.0.tgz",
"integrity": "sha1-dET9hVHd8+XacZj+oMkbyDCMwnQ=",
"requires": {
"pako": "~0.2.5"
}
},
"node": {
"version": "11.15.0",
"resolved": "https://registry.npmjs.org/node/-/node-11.15.0.tgz",
"integrity": "sha512-Nbzq8qr133iwjGo0ZtzQR0mYeawW2eddYpW/k/+yjgbQW2/zG1a/5QiizVOrJ8yc4hbu3369zkj4dDIEd8f7dg==",
"requires": {
"node-bin-setup": "^1.0.0"
}
},
"node-bin-setup": {
"version": "1.0.6",
"resolved": "https://registry.npmjs.org/node-bin-setup/-/node-bin-setup-1.0.6.tgz",
"integrity": "sha512-uPIxXNis1CRbv1DwqAxkgBk5NFV3s7cMN/Gf556jSw6jBvV7ca4F9lRL/8cALcZecRibeqU+5dFYqFFmzv5a0Q=="
},
"node-zip": {
"version": "1.1.1",
"resolved": "https://registry.npmjs.org/node-zip/-/node-zip-1.1.1.tgz",
"integrity": "sha1-lNGtZ0o81GoViN1zb0qaeMdX62I=",
"requires": {
"jszip": "2.5.0"
}
},
"pako": {
"version": "0.2.9",
"resolved": "https://registry.npmjs.org/pako/-/pako-0.2.9.tgz",
"integrity": "sha1-8/dSL073gjSNqBYbrZ7P1Rv4OnU="
},
"sax": {
"version": "1.2.4",
"resolved": "https://registry.npmjs.org/sax/-/sax-1.2.4.tgz",
"integrity": "sha512-NqVDv9TpANUjFm0N8uM5GxL36UgKi9/atZw+x7YFnQ8ckwFGKrl4xX4yWtrey3UJm5nP1kUbnYgLopqWNSRhWw==",
"dev": true
},
"supports-color": {
"version": "5.5.0",
"resolved": "https://registry.npmjs.org/supports-color/-/supports-color-5.5.0.tgz",
"integrity": "sha512-QjVjwdXIt408MIiAqCX4oUKsgU2EqAGzs2Ppkm4aQYbjm+ZEWEcW4SfFNTr4uMNZma0ey4f5lgLrkB0aX0QMow==",
"dev": true,
"requires": {
"has-flag": "^3.0.0"
}
},
"typescript": {
"version": "3.4.5",
"resolved": "https://registry.npmjs.org/typescript/-/typescript-3.4.5.tgz",
"integrity": "sha512-YycBxUb49UUhdNMU5aJ7z5Ej2XGmaIBL0x34vZ82fn3hGvD+bgrMrVDpatgz2f7YxUMJxMkbWxJZeAvDxVe7Vw=="
},
"xml2js": {
"version": "0.4.19",
"resolved": "https://registry.npmjs.org/xml2js/-/xml2js-0.4.19.tgz",
"integrity": "sha512-esZnJZJOiJR9wWKMyuvSE1y6Dq5LCuJanqhxslH2bxM6duahNZ+HMpCLhBQGZkbX6xRf8x1Y2eJlgt2q3qo49Q==",
"dev": true,
"requires": {
"sax": ">=0.6.0",
"xmlbuilder": "~9.0.1"
}
},
"xmlbuilder": {
"version": "9.0.7",
"resolved": "https://registry.npmjs.org/xmlbuilder/-/xmlbuilder-9.0.7.tgz",
"integrity": "sha1-Ey7mPS7FVlxVfiD0wi35rKaGsQ0=",
"dev": true
}
}
}

27
developer/js/package.json Normal file
View file

@ -0,0 +1,27 @@
{
"name": "@keymanapp/developer-lexical-model-compiler",
"version": "1.0.0",
"description": "Keyman Developer lexical model compiler",
"main": "dist/index.js",
"types": "dist/index.d.ts",
"scripts": {
"build": "npx tsc -p .",
"test": "echo \"Error: no test specified\" && exit 1"
},
"author": "Marc Durdin (SIL)",
"license": "MIT",
"bugs": {
"url": "https://github.com/keymanapp/keyman/issues"
},
"homepage": "https://github.com/keymanapp/keyman#readme",
"dependencies": {
"node": "^11.7.0",
"node-zip": "^1.1.1",
"typescript": "^3.2.4"
},
"devDependencies": {
"@types/node": "^10.14.6",
"chalk": "^2.4.2",
"xml2js": "^0.4.19"
}
}

View file

@ -0,0 +1,16 @@
{
"compilerOptions": {
"target": "es6",
"module": "commonjs",
"outDir": "dist",
"sourceMap": true,
"declaration": true,
"typeRoots": [
"./node_modules/@types"
]
},
"exclude": [
"node_modules",
"dist"
]
}