diff --git a/HISTORY.md b/HISTORY.md index 759b782f9a..dbace5ac2e 100644 --- a/HISTORY.md +++ b/HISTORY.md @@ -1,5 +1,24 @@ # Keyman Version History +## 18.0.69 alpha 2024-07-05 + +* fix(core): allow to successfully build on Ubuntu 24.04 (#11926) +* chore(windows): correct output file for 64-bit build of keyman32 in build.sh (#11930) +* chore(android,ios): Add Crowdin localization for Polytonic Greek (#11877) + +## 18.0.68 alpha 2024-07-04 + +* refactor(windows): merge keyman64 build into keyman32 (#11906) +* refactor(windows): remove wm_keyman_keydown and wm_keyman_keyup (#11920) + +## 18.0.67 alpha 2024-07-03 + +* refactor(common/models): move TS priority-queue implementation to web-utils (#11867) + +## 18.0.66 alpha 2024-07-02 + +* fix(developer): handle second parameter of index correctly in kmcmplib compiler (#11815) + ## 18.0.65 alpha 2024-07-01 * fix(developer): prevent non-BMP characters in key part of rule (#11806) diff --git a/VERSION.md b/VERSION.md index fbc6447e6d..0427a9f66d 100644 --- a/VERSION.md +++ b/VERSION.md @@ -1 +1 @@ -18.0.66 \ No newline at end of file +18.0.70 \ No newline at end of file diff --git a/android/KMAPro/kMAPro/src/main/res/values-b+el/strings.xml b/android/KMAPro/kMAPro/src/main/res/values-b+el/strings.xml new file mode 100644 index 0000000000..43e43eb4ca --- /dev/null +++ b/android/KMAPro/kMAPro/src/main/res/values-b+el/strings.xml @@ -0,0 +1,164 @@ + + + + + Μοιρασθῆτε + + Φυλλομετρητής + + Μέγεθος κειμένου + + Περισσότερα + + Διαγραφὴ κειμένου + + Πληροφορίες + + Ρυθμίσεις + + Εγκατάσταση Ενημερώσεων + + Ἔκδοση %1$s + + Τὸ Keyman ἀπαιτεῖ ἔκδοση Chrome 57 ἢ νεώτερη. + + Ἐνημερῶστε τὸ Chrome + + Ἀρχίστε νὰ γράφετε ἐδῶ… + + + + Μέγεθος κειμένου: %1$d + + Μεγαλῶστε τὸ κείμενο + + Μεγαλῶστε τὸ κείμενο + + Ρύθμιση μεγέθους κειμένου + + \nΤὸ κείμενο θὰ ἐκκαθαρισθεῖ πλήρως\n + + Μικρύνετε τὸ κείμενο + + Προσθέστε πληκτρολόγιο γιὰ τὴν γλῶσσα σας + + Ὁρίστε τὸ Κῆμαν ὡς παν-συστημικὸ πληκτρολόγιο + + Ὁρίστε τὸ Κῆμαν ὡς προεπιλεγμένο πληκτρολόγιο + + Περισσότερες πληροφορίες + + Κατὰ τὴν ἔναρξη νὰ προβάλλεται τὸ \"%1$s + + Γιὰ νὰ ἐγκαταστήσετε πακέτα πληκτρολογίων, ἐπιτρέψτε στὸ Κῆμαν νὰ διαβάζει ἐξωτερικὸ χῶρο ἀποθήκευσης. + + Ἀπερρίφθη αἴτημα χώρου ἀποθήκευσης. Πιθανὴ ἀποτυχία ἐγκατάστασης πακέτου πληκτρολογίου + Ἀπερρίφθη αἴτημα χώρου ἀποθηκεύσεως. Δοκιμάστε τὶς ρυθμίσεις Κῆμαν - Ἐγκατάσταση ἀπὸ τοπικὸ ἀρχεῖο + + Ρυθμίσεις + + + Ἐγκατεστημένες γλῶσσες (%1$d) + Ἐγκατεστημένες γλῶσσες (%1$d) + + + Ἐγκαταστῆστε πληκτρολόγιο ἢ λεξικό + + Γλῶσσα προβολῆς + + Μεταβολὴ ὕψους πληκτρολογίου + + Spacebar caption + + Πληκτρολόγιο + + Γλῶσσα + + Γλῶσσα + Πληκτρολόγιο + + Κενό + + \'Ονομα πληκτρολογίου στὸ πλῆκτρο διαστήματος + + \'Ονομα γλώσσας στὸ πλῆκτρο διαστήματος + + \'Ονομα πληκτρολογίου καὶ γλώσσας στὸ πλῆκτρο διαστήματος + + Καμμία λεζάντα στὸ πλῆκτρο διαστήματος + + Δὸνηση κατὰ τὴν πληκτρολόγηση + + Νὰ ἐμφανίζεται πάντα banner + + Πρὸς ὑλοποίησιν + + Ὅταν εἶναι off, ἐμφανίζεται μόνο ὅταν ἔχει ἐνεργοποιηθεῖ τὸ προγνωστικὸ κείμενο + + Ἐπιτρέψτε τὴν ἀποστολὴ ἀναφορῶν κατάρρευσης μέσῳ δικτύου + + Ὅταν εἶναι ΟΝ, θὰ ἀποστέλλονται ἀναφορὲς κατάρρευσης + + Ὅταν εἶναι off, δὲν θὰ ἀποστέλλονται ἀναφορὲς κατάρρευσης + + Ἐγκατάσταση ἀπὸ τὸ keyman.com + + Ἐγκατάσταση ἀπὸ τοπικὸ ἀρχεῖο + + Ἐγκατάσταση ἀπὸ ἄλλη συσκευή + + Προσθέστε γλῶσσες σὲ ἐγκατεστημένο πληκτρολόγιο + + (ἀπὸ πακέτο πληκτρολογίου) + + Ἐπιλέξτε Πακέτο Πληκτρολογίου + + Ἐπιλέξτε γλῶσσες γιὰ τὸ %1$s + + Προσετέθη ἡ γλῶσσα %1$s στὸ %2$s + + Ὅλες οἱ γλῶσσες ἔχουν ἤδη ἐγκατασταθεῖ + + Σύρετε τὸ πληκτρολόγιο γιὰ νὰ ἀλλάξετε τὸ ὕψος + + Περιστρέψτε τὴν συσκευὴ γιὰ λειτουργία πορτραίτου καὶ τοπίου + + Ἐπαναφέρετε τὶς προεπιλεγμένες ρυθμίσεις + + Ἀναζητῆστε ἢ πληκτρολογῆστε URL + + Σελιδοδεῖκτες + + Δὲν ὑπάρχουν σελιδοδεῖκτες + + Προσθέστε σελιδοδείκτη + + Τίτλος + + URL + + Τὸ πακέτο %1$s ἀπέτυχε νὰ ἐγκατασταθεῖ + + Λήψη πακέτου πληκτρολογίου\n%1$s… + + Ἀποτυχία ἐξαγωγῆς + + Ἐγκαταστῆστε πληκτρολόγιο + + Ἐγκαταστῆστε Λεξικό + + Τὸ %1$s δὲν εἶναι ἔγκυρο ἀρχεῖο πακέτου Κῆμαν.\n%2$s\" + + Τὸ πακέτο πληκτρολογίου δὲν ἔχει βελτιστοποιημένα πληκτρολόγια ἀφῆς πρὸς ἐγκατάστασιν + + Δὲν ὑπάρχει νέο προγνωστικό κείμενο πρὸς ἐγκατάστασιν + + Δὲν ὑπάρχουν πληκτρολόγια ἢ προγνωστικό κείμενο πρὸς ἐγκατάστασιν + + Τὸ πακέτο πληκτρολογίου δὲν ἔχει σχετικὲς μὲ αὐτὸ γλῶσσες πρὸς ἐγκατάστασιν + + Ἄκυρα/ἐλλιπῆ μεταδεδομένα στὸ πακέτο + + Τὸ πληκτρολόγιο ἀπαιτεῖ νεώτερη ἔκδοση τοῦ Κῆμαν + + Ἀδυναμία ἐκκινήσεως φυλλομετρητῆ + diff --git a/android/KMEA/app/src/main/java/com/keyman/engine/DisplayLanguages.java b/android/KMEA/app/src/main/java/com/keyman/engine/DisplayLanguages.java index 27c462e95a..6a159523a2 100644 --- a/android/KMEA/app/src/main/java/com/keyman/engine/DisplayLanguages.java +++ b/android/KMEA/app/src/main/java/com/keyman/engine/DisplayLanguages.java @@ -62,6 +62,7 @@ public class DisplayLanguages { new DisplayLanguageType("nl-NL", "Nederlands (Dutch)"), new DisplayLanguageType("ann", "Obolo"), new DisplayLanguageType("pl-PL", "Polski (Polish)"), + new DisplayLanguageType("el", "Polytonic Greek"), new DisplayLanguageType("pt-PT", "Português do Portugal"), new DisplayLanguageType("ff-ZA", "Pulaar-Fulfulde"), // or Fulah new DisplayLanguageType("ru-RU", "Pyccĸий (Russian)"), diff --git a/android/KMEA/app/src/main/res/values-b+el/strings.xml b/android/KMEA/app/src/main/res/values-b+el/strings.xml new file mode 100644 index 0000000000..871171d63f --- /dev/null +++ b/android/KMEA/app/src/main/res/values-b+el/strings.xml @@ -0,0 +1,172 @@ + + + + + + + Πληκτρολόγιο + Πληκτρολόγια + + + + Ἄλλη μέθοδος εἰσαγωγῆς + Ἄλλες Μέθοδοι Εἰσαγωγῆς + + + Προσθέστε νέο Πληκτρολόγιο + + Ἐγκατεστημένες γλῶσσες + + Ρυθμίσεις %1$s + + Προσθέστε + + Πίσω + + Ἀκυρῶστε + + Κλεῖστε + + Κλεῖστε τὸ Κῆμαν + + Προχωρῆστε + + Ἑπόμενη Μέθοδος Εἰσαγωγῆς + + Λήψη + + Ἐγκαταστῆστε + + Ἀργότερα + + Ἑπόμενο + + OK + + Ἐνημερῶστε + + Δὲν ὑπάρχει σύνδεση μὲ τὸ Διαδίκτυο + + Ἀδυναμία συνδέσεως μὲ τὸν διακομιστὴ τοῦ Κῆμαν! + + Θέλετε νὰ διαγράψετε αὐτὸ τὸ πληκτρολόγιο; + + Θὰ θέλατε νὰ κατεβάσετε τὴν τελευταία ἔκδοση αὐτοῦ τοῦ πληκτρολογίου; + + Θά θέλατε νὰ ἐνημερώσετε τώρα πληκτρολόγια καὶ λεξικά; + + Ἐνημερώσεις Πόρων + + Διαθέσιμες Ἐνημερώσεις Πόρων + + %1$s (Διαθέσιμη Ἐνημέρωση) + + Διαθέσιμες ἐνημερώσεις γιὰ τὸ πληκτρολόγιο %1$s: %2$s + + Διαθέσιμες ἐνημερώσεις γιὰ τὸ λεξικὸ %1$s: %2$s + + Ἔκδοση πληκτρολογίου + + Σύνδεσμος βοηθείας + + Ἀπεγκαταστῆστε πληκτρολόγιο + + [νέο] %1$s + + Σαρῶστε αὐτὸν τὸν κωδικό γιὰ νὰ φορτώσετε\nαὐτὸ τὸ πληκτρολόγιο σὲ ἄλλη συσκευή + + Καλωσορίσατε στὸ %1$s + + Ἀπαιτεῖται βιβλιοθήκη FileProvider γιὰ νὰ δεῖτε ἀρχεῖο βοηθείας: %1$s + + Μοιραῖο σφάλμα πληκτρολογίου στὸ %1$s:%2$s γιὰ τὴν %3$s γλῶσσα. Φορτώνεται προεπιλεγμένο πληκτρολόγιο. + + Error in keyboard %1$s:%2$s for %3$s language. + + Ἔλεγχος συσχετισμένου λεξικοῦ πρὸς λῆψιν + Ἀδυναμία συνδέσεως μὲ τὸν διακομιστὴ Κῆμαν γιὰ τὸν ἔλεγχο συσχετισμένου λεξικοῦ πρὸς λῆψιν + + Θὰ θέλατε νὰ κατεβάσετε τὴν τελευταία ἔκδοση αὐτοῦ τοῦ λεξικοῦ; + + Δὲν ὑπάρχει λεξικὸ πρὸς λῆψιν + + Μὴ διαθέσιμος κατάλογος πόρων + + Ἔχει ξεκινήσει ἐνημέρωση καταλόγου στὸ παρασκήνιο + + Ἡ λήψη τοῦ καταλόγου συνεχίζεται· παρακαλοῦμε ξαναδοκιμάστε σὲ λίγο! + + Ἔλεγχος πόρου σὲ ἐξέλιξη + + Ἡ λήψη τοῦ πληκτρολογίου ἔχει ξεκινήσει στὸ παρασκήνιο + + Ἡ λήψη τοῦ ἐπιλεγέντος πληκτρολογίου βρίσκεται σὲ ἐξέλιξη· παρακαλοῦμε ξαναδοκιμάστε σὲ λίγο! + + Ἡ λήψη τοῦ πληκτρολογίου ὁλοκληρώθηκε! + + Ἡ λήψη τοῦ λεξικοῦ ἔχει ξεκινήσει στὸ παρασκήνιο + + Ἡ λήψη τοῦ ἐπιλεγέντος λεξικοῦ βρίσκεται σὲ ἐξέλιξη· παρακαλοῦμε ξαναδοκιμάστε σὲ λίγο! + + Ἡ λήψη τοῦ λεξικοῦ ὁλοκληρώθηκε. + + Ἡ λήψη ἀπέτυχε + + Ἀποτυχία ἀνακτήσεως ληφθέντος ἀρχείου + + Ἀποτυχία προσβάσεως στὸν διακομιστή! + + "Ὅλοι οἱ πόροι ἔχουν ἐνημερωθεῖ!" + + Ἕνας ἢ περισσότεροι πόροι ἀπέτυχαν νὰ ἐνημερωθοῦν! + + Οἱ πόροι ἐνημερώθηκαν ἐπιτυχῶς! + + Ἔκδοση λεξικοῦ + + Ἀπεγκαταστῆτε λεξικό + + Θὰ θέλατε νὰ διαγράψετε αὐτὸ τὸ λεξικό; + + Τὸ λεξικὸ διεγράφη + + Τὸ πληκτρολόγιο %1$s ἐγκατεστάθη + + Τὸ πληκτρολόγιο διεγράφη + + Ἐνεργοποιῆστε τὶς διορθώσεις + + Ἐνεργοποιῆστε προβλέψεις + + Λεξικά + + Λεξικό + Λεξικά + + + Ἔλεγχος διαθεσίμου λεξικοῦ + Ἔλεγχος λεξικῶν ὀνλάϊν + + Λεξικό: %1$s + + %1$s λεξικά + + + Τὸ λεξικὸ ἐγκατεστάθη + + + (%1$d πληκτρολόγιο) + (%1$d πληκτρολόγια) + + + Προεπιλεγμένη Γλῶσσα + + + + Διαγράψτε + + + Κτυπῆστε ἐδῶ γιὰ νὰ ἀλλάξετε πληκτρολόγιο + + Ἀδυναμία ἐκκινήσεως φυλλομετρητῆ + diff --git a/common/models/templates/src/index.ts b/common/models/templates/src/index.ts index 52650ba9bb..8563004fef 100644 --- a/common/models/templates/src/index.ts +++ b/common/models/templates/src/index.ts @@ -2,7 +2,6 @@ export { SENTINEL_CODE_UNIT, applyTransform, buildMergedTransform, isHighSurrogate, isLowSurrogate, isSentinel, transformToSuggestion, defaultApplyCasing } from "./common.js"; -export { default as PriorityQueue, Comparator } from "./priority-queue.js"; export { default as QuoteBehavior } from "./quote-behavior.js"; export { Tokenization, tokenize, getLastPreCaretToken, wordbreak } from "./tokenization.js"; export { default as TrieModel, TrieModelOptions } from "./trie-model.js"; \ No newline at end of file diff --git a/common/models/templates/src/trie-model.ts b/common/models/templates/src/trie-model.ts index 0a571517c7..2cac0429c0 100644 --- a/common/models/templates/src/trie-model.ts +++ b/common/models/templates/src/trie-model.ts @@ -26,12 +26,11 @@ // Should probably make a 'lm-utils' submodule. // Allows the kmwstring bindings to resolve. -import { extendString } from "@keymanapp/web-utils"; +import { extendString, PriorityQueue } from "@keymanapp/web-utils"; import { default as defaultWordBreaker } from "@keymanapp/models-wordbreakers"; import { applyTransform, isHighSurrogate, isSentinel, SENTINEL_CODE_UNIT, transformToSuggestion } from "./common.js"; import { getLastPreCaretToken } from "./tokenization.js"; -import PriorityQueue from "./priority-queue.js"; extendString(); @@ -74,15 +73,6 @@ export interface TrieModelOptions { punctuation?: LexicalModelPunctuation; } -/** - * Used to determine the probability of an entry from the trie. - */ -type TextWithProbability = { - text: string; - // TODO: use negative-log scaling instead? - p: number; // real-number weight, from 0 to 1 -} - class Traversal implements LexiconTraversal { /** * The lexical prefix corresponding to the current traversal state. @@ -95,14 +85,75 @@ class Traversal implements LexiconTraversal { */ root: Node; - constructor(root: Node, prefix: string) { + /** + * The max weight for the Trie being 'traversed'. Needed for probability + * calculations. + */ + totalWeight: number; + + constructor(root: Node, prefix: string, totalWeight: number) { this.root = root; this.prefix = prefix; + this.totalWeight = totalWeight; } - *children(): Generator<{char: string, traversal: () => LexiconTraversal}> { + child(char: USVString): LexiconTraversal | undefined { + /* + Note: would otherwise return the current instance if `char == ''`. If + such a call is happening, it's probably indicative of an implementation + issue elsewhere - let's signal now in order to catch such stuff early. + */ + if(char == '') { + return undefined; + } + + // Split into individual code units. + let steps = char.split(''); + let traversal: Traversal | undefined = this; + + while(steps.length > 0 && traversal) { + const step: string = steps.shift()!; + traversal = traversal._child(step); + } + + return traversal; + } + + // Handles one code unit at a time. + private _child(char: USVString): Traversal | undefined { + const root = this.root; + const totalWeight = this.totalWeight; + const nextPrefix = this.prefix + char; + + if(root.type == 'internal') { + let childNode = root.children[char]; + if(!childNode) { + return undefined; + } + + return new Traversal(childNode, nextPrefix, totalWeight); + } else { + // root.type == 'leaf'; + const legalChildren = root.entries.filter(function(entry) { + return entry.key.indexOf(nextPrefix) == 0; + }); + + if(!legalChildren.length) { + return undefined; + } + + return new Traversal(root, nextPrefix, totalWeight); + } + } + + *children(): Generator<{char: USVString, traversal: () => LexiconTraversal}> { let root = this.root; + // We refer to the field multiple times in this method, and it doesn't change. + // This also assists minification a bit, since we can't minify when re-accessing + // through `this.`. + const totalWeight = this.totalWeight; + if(root.type == 'internal') { for(let entry of root.values) { let entryNode = root.children[entry]; @@ -120,7 +171,7 @@ class Traversal implements LexiconTraversal { let prefix = this.prefix + entry + lowSurrogate; yield { char: entry + lowSurrogate, - traversal: function() { return new Traversal(internalNode.children[lowSurrogate], prefix) } + traversal: function() { return new Traversal(internalNode.children[lowSurrogate], prefix, totalWeight) } } } } else { @@ -131,7 +182,7 @@ class Traversal implements LexiconTraversal { yield { char: entry, - traversal: function () {return new Traversal(entryNode, prefix)} + traversal: function () {return new Traversal(entryNode, prefix, totalWeight)} } } } else if(isSentinel(entry)) { @@ -143,7 +194,7 @@ class Traversal implements LexiconTraversal { let prefix = this.prefix + entry; yield { char: entry, - traversal: function() { return new Traversal(entryNode, prefix)} + traversal: function() { return new Traversal(entryNode, prefix, totalWeight)} } } } @@ -165,30 +216,41 @@ class Traversal implements LexiconTraversal { } yield { char: nodeKey, - traversal: function() { return new Traversal(root, prefix + nodeKey)} + traversal: function() { return new Traversal(root, prefix + nodeKey, totalWeight)} } }; return; } } - get entries(): string[] { + get entries() { + const entryMapper = (value: Entry) => { + return { + text: value.content, + p: value.weight / this.totalWeight + } + } + if(this.root.type == 'leaf') { let prefix = this.prefix; let matches = this.root.entries.filter(function(entry) { return entry.key == prefix; }); - return matches.map(function(value) { return value.content }); + return matches.map(entryMapper); } else { let matchingLeaf = this.root.children[SENTINEL_CODE_UNIT]; if(matchingLeaf && matchingLeaf.type == 'leaf') { - return matchingLeaf.entries.map(function(value) { return value.content }); + return matchingLeaf.entries.map(entryMapper); } else { return []; } } } + + get p(): number { + return this.root.weight / this.totalWeight; + } } /** @@ -286,7 +348,7 @@ export default class TrieModel implements LexicalModel { } public traverseFromRoot(): LexiconTraversal { - return new Traversal(this._trie['root'], ''); + return this._trie.traverseFromRoot(); } }; @@ -307,11 +369,10 @@ export default class TrieModel implements LexicalModel { type SearchKey = string & { _: 'SearchKey'}; /** - * The priority queue will always pop the most weighted item. There can only - * be two kinds of items right now: nodes, and entries; both having a weight - * attribute. + * The priority queue will always pop the most probable item - be it a Traversal + * state or a lexical entry reached via Traversal. */ -type Weighted = Node | Entry; +type TraversableWithProb = TextWithProbability | LexiconTraversal; /** * A function that converts a string (word form or query) into a search key @@ -367,9 +428,9 @@ interface Entry { * Wrapper class for the trie and its nodes. */ class Trie { - private root: Node; + public readonly root: Node; /** The total weight of the entire trie. */ - private totalWeight: number; + readonly totalWeight: number; /** * Converts arbitrary strings to a search key. The trie is built up of * search keys; not each entry's word form! @@ -382,6 +443,10 @@ class Trie { this.totalWeight = totalWeight; } + public traverseFromRoot(): LexiconTraversal { + return new Traversal(this.root, '', this.totalWeight); + } + /** * Lookups an arbitrary prefix (a query) in the trie. Returns the top 3 * results in sorted order. @@ -390,12 +455,8 @@ class Trie { */ lookup(prefix: string): TextWithProbability[] { let searchKey = this.toKey(prefix); - let lowestCommonNode = findPrefix(this.root, searchKey); - if (lowestCommonNode === null) { - return []; - } - - return getSortedResults(lowestCommonNode, searchKey, this.totalWeight); + let rootTraversal = this.traverseFromRoot().child(searchKey); + return rootTraversal ? getSortedResults(rootTraversal) : []; } /** @@ -403,36 +464,10 @@ class Trie { * @param n How many suggestions, maximum, to return. */ firstN(n: number): TextWithProbability[] { - return getSortedResults(this.root, '' as SearchKey, this.totalWeight, n); + return getSortedResults(this.traverseFromRoot(), n); } } -/** - * Finds the deepest descendent in the trie with the given prefix key. - * - * This means that a search in the trie for a given prefix has a best-case - * complexity of O(m) where m is the length of the prefix. - * - * @param key The prefix to search for. - * @param index The index in the prefix. Initially 0. - */ -function findPrefix(node: Node, key: SearchKey, index: number = 0): Node | null { - // An important note - the Trie itself is built on a per-JS-character basis, - // not on a UTF-8 character-code basis. - if (node.type === 'leaf' || index === key.length) { - return node; - } - - // So, for SMP models, we need to match each char of the supplementary pair - // in sequence. Each has its own node in the Trie. - let char = key[index]; - if (node.children[char]) { - return findPrefix(node.children[char], key, index + 1); - } - - return null; -} - /** * Returns all entries matching the given prefix, in descending order of * weight. @@ -441,72 +476,36 @@ function findPrefix(node: Node, key: SearchKey, index: number = 0): Node | null * @param results the current results * @param queue */ -function getSortedResults(node: Node, prefix: SearchKey, N: number, limit = MAX_SUGGESTIONS): TextWithProbability[] { - let queue = new PriorityQueue(function(a: Weighted, b: Weighted) { +function getSortedResults(traversal: LexiconTraversal, limit = MAX_SUGGESTIONS): TextWithProbability[] { + let queue = new PriorityQueue(function(a: TraversableWithProb, b: TraversableWithProb) { // In case of Trie compilation issues that emit `null` or `undefined` - return (b ? b.weight : 0) - (a ? a.weight : 0); + return (b ? b.p : 0) - (a ? a.p : 0); }); let results: TextWithProbability[] = []; - if (node.type === 'leaf') { - // Assuming the values are sorted, we can just add all of the values in the - // leaf, until we reach the limit. - for (let item of node.entries) { - // String.startsWith is not supported on certain Android (5.0) devices we wish to support. - // Requires a minimum of Chrome 36, as opposed to 5.0's default of 35. - if (item.key.indexOf(prefix) == 0) { - let { content, weight } = item; - results.push({ - text: content, - p: weight / N - }); + queue.enqueue(traversal); - if (results.length >= limit) { - return results; - } - } - } - } else { - queue.enqueue(node); - let next: Weighted | undefined; + while(queue.count > 0) { + const entry = queue.dequeue(); - while (next = queue.dequeue()) { - if (isNode(next)) { - // When a node is next up in the queue, that means that next least - // likely suggestion is among its decsendants. - // So we search all of its descendants! - if (next.type === 'leaf') { - queue.enqueueAll(next.entries); - } else { - // XXX: alias `next` so that TypeScript can be SURE that internal is - // in fact an internal node. Because of the callback binding to the - // original definition of node (i.e., a Node | Entry), this will not - // type-check otherwise. - let internal = next; - queue.enqueueAll(next.values.map(char => { - return internal.children[char]; - })); - } - } else { - // When an entry is up next in the queue, we just add its contents to - // the results! - results.push({ - text: next.content, - p: next.weight / N - }); - if (results.length >= limit) { - return results; - } + if((entry as TextWithProbability)!.text !== undefined) { + const lexicalEntry = entry as TextWithProbability; + results.push(lexicalEntry); + if(results.length >= limit) { + return results; } + } else { + const traversal = entry as LexiconTraversal; + queue.enqueueAll(traversal.entries); + let children: LexiconTraversal[] = [] + for(let child of traversal.children()) { + children.push(child.traversal()); + } + queue.enqueueAll(children); } } + return results; - -} - -/** TypeScript type guard that returns whether the thing is a Node. */ -function isNode(x: Entry | Node): x is Node { - return 'type' in x; } /** diff --git a/common/models/templates/test/test-trie-traversal.js b/common/models/templates/test/test-trie-traversal.js index d3d0496316..09237ee9a0 100644 --- a/common/models/templates/test/test-trie-traversal.js +++ b/common/models/templates/test/test-trie-traversal.js @@ -13,6 +13,12 @@ var smpForUnicode = function(code){ return String.fromCharCode(H, L); } +// Prob: entry weight / total weight +// "the" is the highest-weighted word in the fixture. +const PROB_OF_THE = 1000 / 500500; +const PROB_OF_TRUE = 607 / 500500; +const PROB_OF_TROUBLE = 267 / 500500; + describe('Trie traversal abstractions', function() { it('root-level iteration over child nodes', function() { var model = new TrieModel(jsonFixture('tries/english-1000')); @@ -21,7 +27,11 @@ describe('Trie traversal abstractions', function() { assert.isDefined(rootTraversal); let rootKeys = ['t', 'o', 'a', 'i', 'w', 'h', 'f', 'b', 'n', 'y', 's', 'm', - 'u', 'c', 'd', 'l', 'e', 'j', 'p', 'g', 'v', 'k', 'r', 'q'] + 'u', 'c', 'd', 'l', 'e', 'j', 'p', 'g', 'v', 'k', 'r', 'q']; + + rootKeys.forEach((entry) => assert.isOk(rootTraversal.child(entry))); + assert.isNotOk(rootTraversal.child('x')); + assert.isNotOk(rootTraversal.child('z')); for(let child of rootTraversal.children()) { let keyIndex = rootKeys.indexOf(child.char); @@ -50,6 +60,7 @@ describe('Trie traversal abstractions', function() { assert.isDefined(traversalInner1); assert.isArray(child.traversal().entries); assert.isEmpty(child.traversal().entries); + assert.equal(traversalInner1.p, PROB_OF_THE); for(let tChild of traversalInner1.children()) { if(tChild.char == 'h') { @@ -58,19 +69,28 @@ describe('Trie traversal abstractions', function() { assert.isDefined(traversalInner2); assert.isEmpty(tChild.traversal().entries); assert.isArray(tChild.traversal().entries); + assert.equal(traversalInner2.p, PROB_OF_THE); for(let hChild of traversalInner2.children()) { if(hChild.char == 'e') { eSuccess = true; let traversalInner3 = hChild.traversal(); assert.isDefined(traversalInner3); - assert.isDefined(traversalInner3.entries); - assert.equal(traversalInner3.entries[0], "the"); + assert.deepEqual(traversalInner3.entries, [ + { + text: "the", + p: PROB_OF_THE + } + ]); + assert.equal(traversalInner3.p, PROB_OF_THE); for(let eChild of traversalInner3.children()) { let keyIndex = eKeys.indexOf(eChild.char); assert.notEqual(keyIndex, -1, "Did not find char '" + eChild.char + "' in array!"); + + // THE is not accessible if any of the sub-tries of our 'e' node (traversalInner3). + assert.isBelow(eChild.traversal().p, PROB_OF_THE); eKeys.splice(keyIndex, 1); } } @@ -87,6 +107,38 @@ describe('Trie traversal abstractions', function() { assert.isEmpty(eKeys); }); + it('direct traversal with simple internal nodes', function() { + var model = new TrieModel(jsonFixture('tries/english-1000')); + + let rootTraversal = model.traverseFromRoot(); + assert.isDefined(rootTraversal); + + let eKeys = ['y', 'r', 'i', 'm', 's', 'n', 'o']; + + const tNode = rootTraversal.child('t'); + assert.isOk(tNode); + assert.isDefined(tNode); + assert.isArray(tNode.entries); + assert.isEmpty(tNode.entries); + + const hNode = tNode.child('h'); + assert.isOk(hNode); + assert.isDefined(hNode); + assert.isArray(hNode.entries); + assert.isEmpty(hNode.entries); + + const eNode = hNode.child('e'); + assert.isOk(eNode); + assert.isDefined(eNode); + assert.isArray(eNode.entries); + assert.isNotEmpty(eNode.entries); + assert.equal(eNode.entries[0].text, "the"); + + for(let key of eKeys) { + assert.isOk(eNode.child(key)); + } + }); + it('traversal over compact leaf node', function() { var model = new TrieModel(jsonFixture('tries/english-1000')); @@ -102,6 +154,7 @@ describe('Trie traversal abstractions', function() { assert.isDefined(traversalInner1); assert.isArray(child.traversal().entries); assert.isEmpty(child.traversal().entries); + assert.equal(traversalInner1.p, PROB_OF_THE); for(let tChild of traversalInner1.children()) { if(tChild.char == 'r') { @@ -109,6 +162,7 @@ describe('Trie traversal abstractions', function() { assert.isDefined(traversalInner2); assert.isArray(tChild.traversal().entries); assert.isEmpty(tChild.traversal().entries); + assert.equal(traversalInner2.p, PROB_OF_TRUE); for(let rChild of traversalInner2.children()) { if(rChild.char == 'o') { @@ -137,10 +191,17 @@ describe('Trie traversal abstractions', function() { if(leafChildSequence.length > 0) { assert.isArray(curChild.traversal().entries); assert.isEmpty(curChild.traversal().entries); + assert.equal(curChild.traversal().p, PROB_OF_TROUBLE); } else { let finalTraversal = curChild.traversal(); + assert.equal(finalTraversal.p, PROB_OF_TROUBLE); assert.isDefined(finalTraversal.entries); - assert.equal(finalTraversal.entries[0], 'trouble'); + assert.deepEqual(finalTraversal.entries, [ + { + text: 'trouble', + p: PROB_OF_TROUBLE + } + ]); eSuccess = true; } } while (leafChildSequence.length > 0); @@ -154,7 +215,6 @@ describe('Trie traversal abstractions', function() { assert.isTrue(eSuccess); }); - it('traversal with SMP entries', function() { // Two entries, both of which read "apple" to native English speakers. // One solely uses SMP characters, the other of which uses a mix of SMP and standard. @@ -179,18 +239,20 @@ describe('Trie traversal abstractions', function() { for(let child of rootTraversal.children()) { if(child.char == smpA) { aSuccess = true; - let traversalInner1 = child.traversal(); + const traversalInner1 = child.traversal(); assert.isDefined(traversalInner1); - assert.isArray(child.traversal().entries); - assert.isEmpty(child.traversal().entries); + assert.isArray(traversalInner1.entries); + assert.isEmpty(traversalInner1.entries); + assert.equal(traversalInner1.p, 0.5); // The two entries are equally weighted. for(let aChild of traversalInner1.children()) { if(aChild.char == smpP) { pSuccess = true; - let traversalInner2 = aChild.traversal(); + const traversalInner2 = aChild.traversal(); assert.isDefined(traversalInner2); - assert.isArray(aChild.traversal().entries); - assert.isEmpty(aChild.traversal().entries); + assert.isArray(traversalInner2.entries); + assert.isEmpty(traversalInner2.entries); + assert.equal(traversalInner2.p, 0.5); for(let pChild of traversalInner2.children()) { let keyIndex = pKeys.indexOf(pChild.char); @@ -198,10 +260,11 @@ describe('Trie traversal abstractions', function() { pKeys.splice(keyIndex, 1); if(pChild.char == 'p') { // We'll test traversal with the 'mixed' entry from here. - let traversalInner3 = pChild.traversal(); + const traversalInner3 = pChild.traversal(); assert.isDefined(traversalInner3); - assert.isArray(pChild.traversal().entries); - assert.isEmpty(pChild.traversal().entries); + assert.isArray(traversalInner3.entries); + assert.isEmpty(traversalInner3.entries); + assert.equal(traversalInner3.p, 0.5); // Now to handle the rest, knowing it's backed by a leaf node. let curChild = pChild; @@ -227,12 +290,20 @@ describe('Trie traversal abstractions', function() { // Conditional test - if that was not the final character, entries should be undefined. if(leafChildSequence.length > 0) { - assert.isArray(curChild.traversal().entries); - assert.isEmpty(curChild.traversal().entries); + const nextTraversal = curChild.traversal() + assert.isArray(nextTraversal.entries); + assert.isEmpty(nextTraversal.entries); + assert.equal(nextTraversal.p, 0.5); } else { let finalTraversal = curChild.traversal(); assert.isDefined(finalTraversal.entries); - assert.equal(finalTraversal.entries[0], smpA + smpP + 'pl' + smpE); + assert.deepEqual(finalTraversal.entries, [ + { + text: smpA + smpP + 'pl' + smpE, + p: 1/2 + } + ]); + assert.equal(finalTraversal.p, 0.5); eSuccess = true; } } while (leafChildSequence.length > 0); @@ -249,4 +320,52 @@ describe('Trie traversal abstractions', function() { assert.isEmpty(pKeys); }); + + it('direct traversal with SMP entries', function() { + // Two entries, both of which read "apple" to native English speakers. + // One solely uses SMP characters, the other of which uses a mix of SMP and standard. + var model = new TrieModel(jsonFixture('tries/smp-apple')); + + let rootTraversal = model.traverseFromRoot(); + assert.isDefined(rootTraversal); + + let smpA = smpForUnicode(0x1d5ba); + let smpP = smpForUnicode(0x1d5c9); + let smpL = smpForUnicode(0x1d5c5); + let smpE = smpForUnicode(0x1d5be); + + // Just to be sure our utility function is working right. + assert.equal(smpA + smpP + 'pl' + smpE, "𝖺𝗉pl𝖾"); + + let pKeys = ['p', smpP]; + let leafChildSequence = ['l', smpE]; + + const aNode = rootTraversal.child(smpA); + assert.isOk(aNode); + assert.isNotOk(rootTraversal.child('a')); + + const pNode1 = aNode.child(smpP); + assert.isOk(pNode1); + assert.isNotOk(aNode.child('p')); + + const pNode2 = pNode1.child('p'); + assert.isOk(pNode2); + assert.isOk(pNode1.child(smpP)); // Both exist for this step. + + const lNode = pNode2.child('l'); + assert.isOk(lNode); + assert.isNotOk(pNode2.child(smpL)); + + const eNode = lNode.child(smpE); + assert.isOk(eNode); + assert.isNotOk(lNode.child('e')); + + assert.deepEqual(eNode.entries, [ + { + text: smpA + smpP + 'pl' + smpE, + p: 1/2 + } + ]); + assert.equal(eNode.p, 0.5); + }); }); diff --git a/common/models/types/index.d.ts b/common/models/types/index.d.ts index 30aba3ba3d..0d6223bc64 100644 --- a/common/models/types/index.d.ts +++ b/common/models/types/index.d.ts @@ -19,13 +19,32 @@ declare type USVString = string; declare type CasingForm = 'lower' | 'initial' | 'upper'; +/** + * Represents one lexical entry and its probability.. + */ +type TextWithProbability = { + /** + * A lexical entry (word) offered by the model. + * + * Note: not the search-term keyed part. This will match the actual, unkeyed form. + */ + text: string; + + /** + * The probability of the lexical entry, directly based upon its frequency. + * + * A real-number weight, from 0 to 1. + */ + p: number; +} + /** * Used to facilitate edit-distance calculations by allowing the LMLayer to * efficiently search the model's lexicon in a Trie-like manner. */ declare interface LexiconTraversal { /** - * Provides an iterable pattern used to search for words with a prefix matching + * Provides an iterable pattern used to search for words with a 'keyed' prefix matching * the current traversal state's prefix when a new character is appended. Iterating * across `children` provides 'breadth' to a lexical search. * @@ -50,6 +69,20 @@ declare interface LexiconTraversal { */ children(): Generator<{char: USVString, traversal: () => LexiconTraversal}>; + /** + * Allows direct access to the traversal state that results when appending one + * or more codepoints encoded in UTF-16 to the current traversal state's prefix. + * This allows bypassing iteration among all legal child Traversals. + * + * If such a traversal state is not supported, returns `undefined`. + * + * Note: traversals navigate and represent the lexicon in its "keyed" state, + * as produced by use of the search-term keying function defined for the model. + * That is, if a model "keys" `è` to `e`, there will be no `è` child. + * @param char + */ + child(char: USVString): LexiconTraversal | undefined; + /** * Any entries directly keyed by the currently-represented lookup prefix. Entries and * children may exist simultaneously, but `entries` must always exist when no children are @@ -70,7 +103,14 @@ declare interface LexiconTraversal { * - prefix of 'crepe': ['crêpe', 'crêpé'] * - other examples: https://www.thoughtco.com/french-accent-homographs-1371072 */ - entries: USVString[]; + entries: TextWithProbability[]; + + // Note: `p`, not `maxP` - we want to see the same name for `this.entries.p` and `this.p` + /** + * Gives the probability of the highest-frequency lexical entry that is either a member or + * descendent of the represented trie `Node`. + */ + p: number; } /** @@ -294,6 +334,11 @@ declare interface Suggestion { * to the input text. Ex: 'keep', 'emoji', 'correction', etc. */ tag?: SuggestionTag; + + /** + * Set to true if this suggestion is a valid auto-accept target. + */ + autoAccept?: boolean } interface Reversion extends Suggestion { diff --git a/common/predictive-text/unit_tests/in_browser/cases/worker-dummy-integration.spec.ts b/common/predictive-text/unit_tests/in_browser/cases/worker-dummy-integration.spec.ts index a0f7d77290..2ac5f62d00 100644 --- a/common/predictive-text/unit_tests/in_browser/cases/worker-dummy-integration.spec.ts +++ b/common/predictive-text/unit_tests/in_browser/cases/worker-dummy-integration.spec.ts @@ -36,6 +36,18 @@ describe('LMLayer using dummy model', function () { // Since Firefox can't do JSON imports quite yet. const hazelFixture = await fetch(new URL(`${domain}/resources/json/models/future_suggestions/i_got_distracted_by_hazel.json`)); hazelModel = await hazelFixture.json(); + hazelModel = hazelModel.map((set) => set.map((entry) => { + return { + ...entry, + // Dummy-model predictions all claim probability 1; there's no actual probability stuff + // used here. + 'lexical-p': 1, + // We're predicting from a single transform, not a distribution, so probability 1. + 'correction-p': 1, + // Multiply 'em together. + p: 1, + } + })); }); describe('Prediction', function () { diff --git a/common/test/resources/model-helpers.mjs b/common/test/resources/model-helpers.mjs index 1de10bdb15..b2ad083f96 100644 --- a/common/test/resources/model-helpers.mjs +++ b/common/test/resources/model-helpers.mjs @@ -113,7 +113,18 @@ export function randomToken() { } export function iGotDistractedByHazel() { - return jsonFixture('models/future_suggestions/i_got_distracted_by_hazel'); + return jsonFixture('models/future_suggestions/i_got_distracted_by_hazel').map((set) => set.map((entry) => { + return { + ...entry, + // Dummy-model predictions all claim probability 1; there's no actual probability stuff + // used here. + 'lexical-p': 1, + // We're predicting from a single transform, not a distribution, so probability 1. + 'correction-p': 1, + // Multiply 'em together. + p: 1, + } + })); } export function jsonFixture(name, root, import_root) { diff --git a/common/web/input-processor/src/text/prediction/languageProcessor.ts b/common/web/input-processor/src/text/prediction/languageProcessor.ts index a24938fe27..921f4abf3c 100644 --- a/common/web/input-processor/src/text/prediction/languageProcessor.ts +++ b/common/web/input-processor/src/text/prediction/languageProcessor.ts @@ -19,7 +19,7 @@ export type StateChangeHandler = (state: StateChangeEnum) => any; /** * Covers 'tryaccept' events. */ -export type TryUIHandler = (source: string) => boolean; +export type TryUIHandler = (source: string, returnObj: {shouldSwallow: boolean}) => boolean; export type InvalidateSourceEnum = 'new'|'context'; @@ -168,9 +168,9 @@ export default class LanguageProcessor extends EventEmitter; private swallowPrediction: boolean = false; @@ -191,7 +192,7 @@ export default class PredictionContext extends EventEmitter { - //let keyman = com.keyman.singleton; + private doTryAccept = (source: string, returnObj: {shouldSwallow: boolean}): void => { + const recentAcceptCause = this.recentAcceptCause; - if(!this.recentAccept && this.selected) { + if(!recentAcceptCause && this.selected) { this.accept(this.selected); - // returnObj.shouldSwallow = true; - } else if(this.recentAccept && source == 'space') { - this.recentAccept = false; - // // If the model doesn't insert wordbreaks, don't swallow the space. If it does, - // // we consider that insertion to be the results of the first post-accept space. - // returnObj.shouldSwallow = !!keyman.core.languageProcessor.wordbreaksAfterSuggestions; // can be handed outside + returnObj.shouldSwallow = true; + + // doTryAccept is the path for keystroke-based auto-acceptance. + // Overwrite the cause to reflect this. + this.recentAcceptCause = 'key'; + } else if(recentAcceptCause && source == 'space') { + this.recentAcceptCause = null; + if(recentAcceptCause == 'key') { + // No need to swallow the keystroke's whitespace; we triggered the prior acceptance + // FROM a space, so we've already aliased the suggestion's built-in space. + returnObj.shouldSwallow = false; + return; + } + + // Standard whitespace applications from the banner, those we DO want to + // swallow the first time. + // + // If the model doesn't insert wordbreaks, there's no space to alias, so + // don't swallow the space. If it does, we consider that insertion to be + // the results of the first post-accept space. + returnObj.shouldSwallow = !!this.langProcessor.wordbreaksAfterSuggestions; // can be handed outside } else { - // returnObj.shouldSwallow = false; + returnObj.shouldSwallow = false; } } @@ -250,9 +268,9 @@ export default class PredictionContext extends EventEmitter { // By default, we assume that the context is the same until we notice otherwise. this.initNewContext = false; + this.selected = null; if(!this.swallowPrediction || source == 'context') { - this.recentAccept = false; + this.recentAcceptCause = null; this.doRevert = false; this.recentRevert = false; @@ -299,7 +318,7 @@ export default class PredictionContext extends EventEmitter