chore(web): Merge branch 'master' into feat/web/gesture-recognizer-main

This commit is contained in:
Joshua A. Horton 2022-09-19 12:49:51 +07:00
commit 53cddf71e3
137 changed files with 3027 additions and 1817 deletions

42
.github/labeler.yml vendored
View file

@ -5,20 +5,19 @@
# common ones. The others are commented out. There is still some variance between
# folder names and labels; consider this documentation of that ;-)
docs: docs/**
#
# Add labels based on changed files using actions/labeler
#
android/: android/**
android/app/: android/KMAPro/**
#android/browser/:
android/engine/: android/KMEA/**
#android/resources/:
android/samples/: android/Samples/**
common/:
- common/**
- core/**
- resources/**
common/models/: common/models/**
@ -29,46 +28,39 @@ common/models/wordbreakers/: common/models/wordbreakers/**
common/resources/: resources/**
common/web/: common/web/**
common/core/:
core/:
- core/**
developer/:
- developer/**
- windows/src/developer/**
developer/compilers/:
- windows/src/developer/kmcomp/**
- developer/src/kmlmc/**
- developer/src/kmcomp/**
- developer/src/kmcmpdll/**
- developer/src/kmc/**
- developer/src/kmc-*/**
developer/ide/:
- windows/src/developer/TIKE/**
- developer/src/server/**
- developer/src/tike/**
# developer/resources/
# developer/tools/
ios/: ios/**
ios/app/: ios/keyman/**
# ios/browser/
ios/engine/: ios/engine/**
# ios/resources/
ios/samples/: ios/samples/**
linux/: linux/**
linux/config/: linux/keyman-config/**
linux/engine/:
- linux/ibus-keyman/**
- linux/ibus-kmfl/**
- linux/kmflcomp/**
- linux/libkmfl/**
- linux/scim_kmfl_imengine/**
# linux/resources/
# linux/samples/
- linux/legacy/ibus-kmfl/**
- linux/legacy/kmflcomp/**
- linux/legacy/libkmfl/**
mac/: mac/**
# mac/config/:
mac/engine/: mac/**
# mac/resources/
# mac/samples/
# mac/engine/: mac/**
oem/: oem/**
oem/fv/: oem/firstvoices/**
@ -79,15 +71,9 @@ oem/fv/windows/: oem/firstvoices/windows/**
web/: web/**
# web/bookmarklet/
web/engine/: web/source/**
# web/resources/
web/ui/: web/source/kmwui*
web/samples/: web/samples/**
# Somewhat messy since we try and exclude Developer :)
windows/:
- any: ['windows/**', '!windows/src/developer/**']
windows/: windows/**
windows/config/: windows/src/desktop/**
windows/engine/: windows/src/engine/**
# windows/resources/
# windows/samples/

View file

@ -11,6 +11,10 @@ version: v1
#
labels:
#
# conventional commit / semantic PR styles
#
- label: 'feat'
matcher:
title: '^feat(\(|:)'
@ -20,26 +24,48 @@ labels:
- label: 'chore'
matcher:
title: '^chore(\(|:)'
# "change",
- label: 'docs'
matcher:
title: '^docs(\(|:)'
# "style",
- label: 'refactor'
matcher:
title: '^refactor(\(|:)'
# "test",
- label: 'auto'
matcher:
title: '^auto(\(|:)'
# Below are the scopes that we look for in the PR title
#
# additional meta flags
#
# stable-targeted patches; note, this does not pick up chained PRs automatically
- label: 'stable'
matcher:
baseBranch: '^stable-.+'
# PRs marked as cherry-picks by title
- label: 'cherry-pick'
matcher:
title: '(🍒|:cherries:)'
# long-lived feature branches
- label: 'feature-branch'
matcher:
branch: '^feature-.+'
#
# Scopes that we look for in the PR title
#
- label: 'android/'
matcher:
title: '\(.*android.*\):'
- label: 'common/'
matcher:
title: '\(.*common.*\):'
- label: 'core/'
matcher:
title: '\(.*core.*\):'
- label: 'developer/'
matcher:
title: '\(.*developer.*\):'
@ -62,6 +88,10 @@ labels:
matcher:
title: '\(.*windows.*\):'
- label: 'cherry-pick'
#
# epics -- we will add/remove these as we work on new epics each release
#
- label: 'epic-ldml'
matcher:
title: '(🍒|:cherries:)'
branch: '.*epic-ldml.*' # anywhere in the branch name, e.g. feat/epic-ldml/developer/... or feat/developer/foo-epic-ldml

View file

@ -12,7 +12,7 @@ jobs:
repo-token: "${{ secrets.GITHUB_TOKEN }}"
- name: Update labels based on PR title
id: labeler
uses: fuxingloh/multi-labeler@8afa186ed03230c98fe24ebf9fe35093072ad46e # v1.4.0
uses: fuxingloh/multi-labeler@fb9bc28b2d65e406ffd208384c5095793c3fd59a # v1.8.0
with:
github-token: ${{secrets.GITHUB_TOKEN}}
config-path: .github/multi-labeler.yml

View file

@ -1,5 +1,105 @@
# Keyman Version History
## 16.0.66 alpha 2022-09-17
* chore: improve auto labeling (#7288)
## 16.0.65 alpha 2022-09-16
* chore(linux): Remove unused IBusLookupTable (#7296)
## 16.0.64 alpha 2022-09-15
* fix(android/engine): Switch keyboard if uninstalling current one (#7291)
* fix(common/models): fixes quote-adjacent pred-text suggestions (#7205)
* fix(common/models): max prediction wait check (#7290)
## 16.0.63 alpha 2022-09-14
* chore(linux): Update debian changelog (#7281)
* fix(linux): Fix ignored error (#7284)
## 16.0.62 alpha 2022-09-13
* fix(developer): hide key-sizes when in desktop layout in touch layout editor (#7225)
* fix(developer): show more useful error if out of space during Setup (#7267)
## 16.0.61 alpha 2022-09-12
* docs(windows): add steps for using testhost debugging (#7263)
* fix(developer): compiler mismatch on currentLine (#7190)
* fix(developer): suppress repeated warnings about unreachable code (#7219)
* chore: try disabling concurrency for browserstack tests (#7258)
* chore(web): disable browserstack on non-web-specific builds (#7260)
## 16.0.60 alpha 2022-09-10
* fix(web): enhanced timer for prediction algorithm (#7037)
* chore(core): fixup km_kbp_event docs (#7253)
* fix(windows) Update unit tests for tsf bkspace (#7254)
* fix(android): Standardize language ID in language picker menu (#7239)
## 16.0.59 alpha 2022-09-09
* chore: use keyman.com instead of keyman-staging.com (#7233)
* feat(core): add `km_kbp_event` API endpoint (#7223)
* fix(windows): Delete both code units when deleting surrogate pairs in TSF-aware apps (#7243)
## 16.0.58 alpha 2022-09-08
* fix(common/models): reconnects unit tests for worker-internal submodules (#7215)
* feat(common/models): extra Unicode-based wordbreaker unit tests (#7217)
## 16.0.57 alpha 2022-09-07
* fix(android/engine): Fix Backspace key to delete without errant subkeys (#7156)
## 16.0.56 alpha 2022-09-05
* chore(linux): Fix ibus-keyman.postinst script (#7192)
## 16.0.55 alpha 2022-09-03
* chore(android): Reduce Toast notification noise (#7178)
## 16.0.54 alpha 2022-08-30
* chore(core): rename json.hpp to jsonpp.hpp (#6993)
* chore(developer): remove unused dependencies from KeymanWeb compiler (#7000)
* chore: update core/ label (#7138)
* chore(linux): Update debian changelog (#7145)
* fix(linux): Allow downgrades for installing build deps (#7147)
* chore(core): emcc off path for linux (#7149)
## 16.0.53 alpha 2022-08-29
* feat(linux): Dockerfile for linux builder (#7133)
## 16.0.52 alpha 2022-08-26
* fix(android/engine): Lower the max height for landscape orientation (#7119)
* fix(linux): Remove wrong `ok_for_single_backspace` method (#7123)
## 16.0.51 alpha 2022-08-24
* fix(android): verify browser before starting activity (#7001)
* chore: Change platform advocates per discussion (#7096)
* fix(windows): Add invalidate context action to non-updatable parse (#7089)
* feat(common): add parameter variable support to `builder_` functions (#7103)
## 16.0.50 alpha 2022-08-23
* chore(core): Remove obsolete python keyboardprocessor (#7094)
## 16.0.49 alpha 2022-08-22
* fix: remove saving and restoring context kbd options (#7077)
* chore(deps): bump @actions/core from 1.8.2 to 1.9.1 (#7087)
## 16.0.48 alpha 2022-08-16
* fix(web): button, float init timer cleanup (#7036)
## 16.0.47 alpha 2022-08-15
* chore(core): refactor kmx_file.h to common (#7062)

View file

@ -1 +1 @@
16.0.48
16.0.67

View file

@ -18,6 +18,7 @@ import android.webkit.WebChromeClient;
import android.webkit.WebSettings;
import android.webkit.WebView;
import android.webkit.WebViewClient;
import android.widget.Toast;
import androidx.appcompat.app.AppCompatActivity;
import com.tavultesoft.kmea.BaseActivity;
@ -113,7 +114,11 @@ public class KMPBrowserActivity extends BaseActivity {
// All links that aren't internal Keyman keyboard links open in user's browser
Intent intent = new Intent(Intent.ACTION_VIEW, uri);
startActivity(intent);
if (intent.resolveActivity(getPackageManager()) != null) {
startActivity(intent);
} else {
Toast.makeText(context, getString(R.string.unable_to_open_browser), Toast.LENGTH_SHORT).show();
}
return true;
}
if (lowerURL.startsWith("keyman:")) {

View file

@ -26,6 +26,7 @@ import android.widget.ListAdapter;
import android.widget.ListView;
import android.widget.SimpleAdapter;
import android.widget.TextView;
import android.widget.Toast;
import com.tavultesoft.kmea.ConfirmDialogFragment;
import com.tavultesoft.kmea.KMHelpFileActivity;
@ -160,7 +161,11 @@ public final class KeyboardSettingsActivity extends AppCompatActivity {
Bundle args = kbd.buildDownloadBundle();
Intent i = new Intent(getApplicationContext(), KMKeyboardDownloaderActivity.class);
i.putExtras(args);
startActivity(i);
if (i.resolveActivity(getPackageManager()) != null) {
startActivity(i);
} else {
Toast.makeText(getApplicationContext(), getString(R.string.unable_to_open_browser), Toast.LENGTH_SHORT).show();
}
finish();
// "Help" link clicked

View file

@ -762,10 +762,24 @@ public class MainActivity extends BaseActivity implements OnKeyboardEventListene
// Launch PlayStore to update Chrome
try {
Intent intent = new Intent(Intent.ACTION_VIEW, Uri.parse("market://details?id=com.android.chrome"));
startActivity(intent);
if (intent.resolveActivity(getPackageManager()) != null) {
startActivity(intent);
} else {
intent = new Intent(Intent.ACTION_VIEW, Uri.parse("https://play.google.com/store/apps/details?id=com.android.chrome"));
if (intent.resolveActivity(getPackageManager()) != null) {
startActivity(intent);
} else {
Toast.makeText(getApplicationContext(), getString(R.string.unable_to_open_browser), Toast.LENGTH_SHORT).show();
}
}
} catch (android.content.ActivityNotFoundException e) {
// Link to Chrome if user is not signed in to Play Store
startActivity(new Intent(Intent.ACTION_VIEW, Uri.parse("https://play.google.com/store/apps/details?id=com.android.chrome")));
Intent intent = new Intent(Intent.ACTION_VIEW, Uri.parse("https://play.google.com/store/apps/details?id=com.android.chrome"));
if (intent.resolveActivity(getPackageManager()) != null) {
startActivity(intent);
} else {
Toast.makeText(getApplicationContext(), getString(R.string.unable_to_open_browser), Toast.LENGTH_SHORT).show();
}
}
}
});
@ -979,10 +993,6 @@ public class MainActivity extends BaseActivity implements OnKeyboardEventListene
String _downloadid = CloudLexicalModelMetaDataDownloadCallback.createDownloadId(languageID);
CloudLexicalModelMetaDataDownloadCallback _callback = new CloudLexicalModelMetaDataDownloadCallback();
Toast.makeText(context,
context.getString(R.string.query_associated_model),
Toast.LENGTH_SHORT).show();
ArrayList<CloudApiTypes.CloudApiParam> aPreparedCloudApiParams = new ArrayList<>();
String url = CloudRepository.prepareLexicalModelQuery(languageID);
aPreparedCloudApiParams.add(new CloudApiTypes.CloudApiParam(

View file

@ -32,6 +32,7 @@ import com.tavultesoft.kmea.util.KMLog;
import org.json.JSONObject;
import java.io.File;
import java.security.InvalidParameterException;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.List;
@ -190,6 +191,33 @@ public final class SelectLanguageFragment extends Fragment implements BlockingSt
listView.setOnItemClickListener(new AdapterView.OnItemClickListener() {
/**
* Utility to modify keyboardList. Based on BCP47.toggleLanguage
* If keyboard's languageID exists in the list, remove the keyboard.
* Otherwise, add the keyboard to the list.
* @param keyboardList
* @param k
*/
private void toggleLanguage(ArrayList<Keyboard> keyboardList, Keyboard k) {
if (keyboardList == null) {
throw new InvalidParameterException("keyboardList must not be null");
}
if (k == null) {
throw new InvalidParameterException("keyboard must not be null");
}
// See if languageID already exists in the keyboardList
for (Keyboard l: keyboardList) {
if (BCP47.languageEquals(l.getLanguageID(), k.getLanguageID())) {
keyboardList.remove(l);
return;
}
}
k.setLanguage(k.getLanguageID().toLowerCase(), k.getLanguageName());
keyboardList.add(k);
}
@Override
public void onItemClick(AdapterView<?> parent, View view, int position, long id) {
Keyboard k = availableKeyboardsList.get(position);
@ -199,21 +227,13 @@ public final class SelectLanguageFragment extends Fragment implements BlockingSt
languageList = new ArrayList<String>();
}
String selectedLanguageID = k.getLanguageID();
if (languageList.contains(selectedLanguageID)) {
languageList.remove(selectedLanguageID);
} else {
languageList.add(selectedLanguageID);
}
BCP47.toggleLanguage(languageList, selectedLanguageID);
} else {
// Otherwise, add the language association
if (addKeyboardsList == null) {
addKeyboardsList = new ArrayList<Keyboard>();
}
if (addKeyboardsList.contains(k)) {
addKeyboardsList.remove(k);
} else {
addKeyboardsList.add(k);
}
toggleLanguage(addKeyboardsList, k);
}
// Disable install button if no languages selected or all languages already installed

View file

@ -247,4 +247,7 @@
<!-- Context: KMP Package strings -->
<string name="minimum_keyboard_version_not_supported" comment="Notification when keyboard has functionality not supported by current Keyman">
Keyboard requires a newer version of Keyman</string>
<!-- Context: anywhere -->
<string name="unable_to_open_browser" comment="Notification when a browser activity cannot be launched">Unable to launch web browser</string>
</resources>

View file

@ -372,8 +372,10 @@ final class KMKeyboard extends WebView {
public void dismissSubKeysWindow() {
try {
if (subKeysWindow != null && subKeysWindow.isShowing())
if (subKeysWindow != null && subKeysWindow.isShowing()) {
subKeysWindow.dismiss();
}
subKeysList = null;
} catch (Exception e) {
KMLog.LogException(TAG, "", e);
}

View file

@ -136,7 +136,11 @@ public final class KeyboardInfoActivity extends BaseActivity {
} else {
Intent i = new Intent(Intent.ACTION_VIEW);
i.setData(Uri.parse(customHelpLink));
startActivity(i);
if (i.resolveActivity(getPackageManager()) != null) {
startActivity(i);
} else {
Toast.makeText(context, getString(R.string.unable_to_open_browser), Toast.LENGTH_SHORT).show();
}
}
}
}

View file

@ -433,7 +433,7 @@ public final class KeyboardPickerActivity extends BaseActivity {
if(adapter != null) {
adapter.notifyDataSetChanged();
}
if (position == curKbPos && listView != null) {
if (position == curKbPos) {
switchKeyboard(0,false);
} else if(listView != null) { // A bit of a hack, since LanguageSettingsActivity calls this method too.
curKbPos = KeyboardController.getInstance().getKeyboardIndex(KMKeyboard.currentKeyboard());

View file

@ -125,31 +125,12 @@ public class CloudLexicalModelMetaDataDownloadCallback implements ICloudDownload
* @param aMetaDataResult the meta data results
*/
private void processCloudResults(Context aContext, List<MetaDataResult> aMetaDataResult) {
for(MetaDataResult _r:aMetaDataResult) {
if (_r.returnjson.target== CloudApiTypes.ApiTarget.Keyboard) {
for (MetaDataResult _r : aMetaDataResult) {
if (_r.returnjson.target == CloudApiTypes.ApiTarget.Keyboard) {
//handleKeyboardMetaData(_r);
}
if(_r.returnjson.target== CloudApiTypes.ApiTarget.KeyboardLexicalModels) {
JSONArray lmData = _r.returnjson.jsonArray;
if (lmData != null && lmData.length() > 0) {
try {
JSONObject modelInfo = lmData.getJSONObject(0);
if (modelInfo.has("packageFilename") && modelInfo.has("id")) {
String _modelID = modelInfo.getString("id");
ArrayList<CloudApiTypes.CloudApiParam> urls = new ArrayList<>();
urls.add(new CloudApiTypes.CloudApiParam(
CloudApiTypes.ApiTarget.LexicalModelPackage,
modelInfo.getString("packageFilename")));
_r.additionalDownloadid = CloudLexicalPackageDownloadCallback.createDownloadId(_modelID);
_r.additionalDownloads= urls;
}
} catch (JSONException e) {
KMLog.LogException(TAG, "Error parsing lexical model from api.keyman.com. ", e);
}
} else {
BaseActivity.makeToast(aContext, R.string.no_associated_model, Toast.LENGTH_SHORT);
}
if (_r.returnjson.target == CloudApiTypes.ApiTarget.KeyboardLexicalModels) {
processCloudResultForModel(aContext, _r);
}
if (_r.returnjson.target == CloudApiTypes.ApiTarget.PackageVersion) {
@ -164,6 +145,32 @@ public class CloudLexicalModelMetaDataDownloadCallback implements ICloudDownload
}
}
private void processCloudResultForModel(Context aContext, MetaDataResult _r) {
JSONArray lmData = _r.returnjson.jsonArray;
if (lmData == null || lmData.length() == 0) {
KMLog.LogError(TAG, "Error in lexical model metadata from api.keyman.com - zero or null");
return;
}
try {
JSONObject modelInfo = lmData.getJSONObject(0);
if (!modelInfo.has("packageFilename") || !modelInfo.has("id")) {
KMLog.LogError(TAG, "Error in lexical model metadata from api.keyman.com - missing metadata");
return;
}
String _modelID = modelInfo.getString("id");
ArrayList<CloudApiTypes.CloudApiParam> urls = new ArrayList<>();
urls.add(new CloudApiTypes.CloudApiParam(
CloudApiTypes.ApiTarget.LexicalModelPackage,
modelInfo.getString("packageFilename")));
_r.additionalDownloadid = CloudLexicalPackageDownloadCallback.createDownloadId(_modelID);
_r.additionalDownloads= urls;
} catch (JSONException e) {
KMLog.LogException(TAG, "Error in lexical model metadata from api.keyman.com. ", e);
}
}
/**
* create a download id for the lexical model metadata.
* @param aLanguageId the language id

View file

@ -39,7 +39,7 @@ public class CloudRepository {
public static final String DOWNLOAD_IDENTIFIER_CATALOGUE = "catalogue";
public static final String API_PRODUCTION_HOST = "api.keyman.com";
public static final String API_STAGING_HOST = "api.keyman-staging.com";
public static final String API_STAGING_HOST = "api.keyman.com"; // #7227 disabling: "api.keyman-staging.com";
public static final String API_MODEL_LANGUAGE_FORMATSTR = "https://%s/model?q=bcp47:%s";
public static final String API_PACKAGE_VERSION_FORMATSTR = "https://%s/package-version?platform=android%s%s";
@ -400,7 +400,6 @@ public class CloudRepository {
BaseActivity.makeToast(context, R.string.catalog_download_is_running_in_background, Toast.LENGTH_SHORT);
} else {
updateIsRunning = true;
BaseActivity.makeToast(context, R.string.catalog_download_start_in_background, Toast.LENGTH_SHORT);
CloudDownloadMgr.getInstance().executeAsDownload(
context, DOWNLOAD_IDENTIFIER_CATALOGUE, memCachedDataset, _download_callback, params);
}

View file

@ -106,7 +106,6 @@ public class ResourcesUpdateTool implements KeyboardEventHandler.OnKeyboardDownl
return;
}
BaseActivity.makeToast(currentContext, R.string.update_check_current, Toast.LENGTH_SHORT);
lastUpdateCheck = Calendar.getInstance();
SharedPreferences prefs = currentContext.getSharedPreferences(currentContext.getString(R.string.kma_prefs_name), Context.MODE_PRIVATE);
SharedPreferences.Editor editor = prefs.edit();

View file

@ -4,6 +4,9 @@
package com.tavultesoft.kmea.util;
import java.util.ArrayList;
import java.security.InvalidParameterException;
public final class BCP47 {
/**
* Utility to compare two language ID strings
@ -18,4 +21,29 @@ public final class BCP47 {
return id1.equalsIgnoreCase(id2);
}
}
/**
* Utility to modify languageList.
* If languageID exists in the list, remove it. Otherwise, add languageID to the list.
* @param languageList
* @param languageID
*/
public static void toggleLanguage(ArrayList<String> languageList, String languageID) {
if (languageList == null) {
throw new InvalidParameterException("languageList must not be null");
}
if (languageID == null) {
throw new InvalidParameterException("languageID must not be null");
}
// See if languageID already exists in the languageList
for (String l: languageList) {
if (languageEquals(l, languageID)) {
languageList.remove(l);
return;
}
}
languageList.add(languageID.toLowerCase());
}
}

View file

@ -21,7 +21,7 @@ import java.util.regex.Pattern;
public final class KMPLink {
public static final String KMP_PRODUCTION_HOST = "keyman.com";
public static final String KMP_STAGING_HOST = "keyman-staging.com";
public static final String KMP_STAGING_HOST = "keyman.com"; // #7227 disabling: "keyman-staging.com";
private static final String KMP_INSTALL_KEYBOARDS_PATTERN_FORMATSTR = "^http(s)?://(%s|%s)/keyboards/install/([^\\?/]+)(\\?(.+))?$";
private static final String installPatternFormatStr = KMString.format(KMP_INSTALL_KEYBOARDS_PATTERN_FORMATSTR,

View file

@ -2,7 +2,7 @@
<!-- Dimens for Landscape (Phone) -->
<resources>
<dimen name="banner_height">40dp</dimen>
<dimen name="keyboard_height">150dp</dimen>
<dimen name="keyboard_height">100dp</dimen>
<dimen name="key_width">49.25dp</dimen>
<dimen name="key_height">49.25dp</dimen>
<dimen name="popup_arrow_width">21dp</dimen>

View file

@ -2,7 +2,7 @@
<!-- Dimens for Landscape (Tablet 7") -->
<resources>
<dimen name="banner_height">50dp</dimen>
<dimen name="keyboard_height">200dp</dimen>
<dimen name="keyboard_height">140dp</dimen>
<dimen name="key_width">73.5dp</dimen>
<dimen name="key_height">58.5dp</dimen>
<dimen name="popup_arrow_width">32dp</dimen>

View file

@ -2,7 +2,7 @@
<!-- Dimens for Landscape (Tablet 10") -->
<resources>
<dimen name="banner_height">60dp</dimen>
<dimen name="keyboard_height">300dp</dimen>
<dimen name="keyboard_height">200dp</dimen>
<dimen name="key_width">98.5dp</dimen>
<dimen name="key_height">98.5dp</dimen>
<dimen name="popup_arrow_width">42dp</dimen>

View file

@ -3,10 +3,10 @@
<!-- Device type -->
<string name="device_type" translatable="false">AndroidMobile</string>
<!-- Application name (Keyman Engine for Android) -->
<string name="app_name" translatable="false">KMEA</string>
<!-- Context: Title for list -->
<plurals name="title_keyboards">
<item quantity="one">Keyboard</item>
@ -267,4 +267,6 @@
<!-- Context: Other strings -->
<string name="help_bubble_text">Tap here to change keyboard</string>
<!-- Context: anywhere -->
<string name="unable_to_open_browser" comment="Notification when a browser activity cannot be launched">Unable to launch web browser</string>
</resources>

View file

@ -18,9 +18,9 @@ public class KMPLinkTest {
Assert.assertFalse(KMPLink.isKeymanInstallLink(""));
// Valid Keyman keyboard install links
Assert.assertTrue(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboards/install/malar_malayalam"));
Assert.assertTrue(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboards/install/malar_malayalam?bcp47=ml"));
Assert.assertTrue(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboards/install/fv_sencoten?bcp47=str-Latn"));
// Assert.assertTrue(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboards/install/malar_malayalam"));
// Assert.assertTrue(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboards/install/malar_malayalam?bcp47=ml"));
// Assert.assertTrue(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboards/install/fv_sencoten?bcp47=str-Latn"));
Assert.assertTrue(KMPLink.isKeymanInstallLink("https://keyman.com/keyboards/install/lao%20unicode")); // legacy keyboard within download dialog
Assert.assertTrue(KMPLink.isKeymanInstallLink("https://keyman.com/keyboards/install/khmer_angkor?_gid=112233&bcp47=lo")); // other query params appearing
@ -28,11 +28,11 @@ public class KMPLinkTest {
Assert.assertTrue(KMPLink.isKeymanInstallLink("https://keyman.com/keyboards/install/malar_malayalam?bcp47=ml"));
// Keyboard link is wrong
Assert.assertFalse(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboard/install/malar_malayalam"));
// Assert.assertFalse(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboard/install/malar_malayalam"));
// link missing packageID
Assert.assertFalse(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboards/install"));
Assert.assertFalse(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboards/install/"));
// Assert.assertFalse(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboards/install"));
// Assert.assertFalse(KMPLink.isKeymanInstallLink("https://keyman-staging.com/keyboards/install/"));
Assert.assertFalse(KMPLink.isKeymanInstallLink("https://keyman.com/keyboards/install"));
Assert.assertFalse(KMPLink.isKeymanInstallLink("https://keyman.com/keyboards/install/"));
}
@ -43,14 +43,14 @@ public class KMPLinkTest {
Assert.assertFalse(KMPLink.isKeymanDownloadLink(""));
// Valid Keyman keyboard download links
Assert.assertTrue(KMPLink.isKeymanDownloadLink("https://keyman-staging.com/go/package/download/malar_malayalam?platform=android&tier=alpha"));
Assert.assertTrue(KMPLink.isKeymanDownloadLink("https://keyman-staging.com/go/package/download/malar_malayalam?platform=android&tier=alpha&bcp47=ml"));
// Assert.assertTrue(KMPLink.isKeymanDownloadLink("https://keyman-staging.com/go/package/download/malar_malayalam?platform=android&tier=alpha"));
// Assert.assertTrue(KMPLink.isKeymanDownloadLink("https://keyman-staging.com/go/package/download/malar_malayalam?platform=android&tier=alpha&bcp47=ml"));
Assert.assertTrue(KMPLink.isKeymanDownloadLink("https://keyman.com/go/package/download/malar_malayalam?platform=android&tier=alpha"));
Assert.assertTrue(KMPLink.isKeymanDownloadLink("https://keyman.com/go/package/download/malar_malayalam?platform=android&tier=alpha&bcp47=ml"));
// link missing packageID
Assert.assertFalse(KMPLink.isKeymanDownloadLink("https://keyman-staging.com/go/package/download"));
Assert.assertFalse(KMPLink.isKeymanDownloadLink("https://keyman-staging.com/go/package/download/"));
// Assert.assertFalse(KMPLink.isKeymanDownloadLink("https://keyman-staging.com/go/package/download"));
// Assert.assertFalse(KMPLink.isKeymanDownloadLink("https://keyman-staging.com/go/package/download/"));
Assert.assertFalse(KMPLink.isKeymanDownloadLink("https://keyman.com/go/package/download"));
Assert.assertFalse(KMPLink.isKeymanDownloadLink("https://keyman.com/go/package/download/"));
}

View file

@ -79,6 +79,22 @@ public class MainActivity extends AppCompatActivity implements OnKeyboardEventLi
KMManager.KMDefault_KeyboardFont,
KMManager.KMDefault_KeyboardFont);
KMManager.addKeyboard(this, platformtestKBbInfo);
// Final K_ENTER test keyboard
Keyboard finalKBInfo = new Keyboard(
"final",
"final",
"final Keyboard",
"en",
"English",
"1.0",
"",
"",
true,
KMManager.KMDefault_KeyboardFont,
KMManager.KMDefault_KeyboardFont);
KMManager.addKeyboard(this, finalKBInfo);
}
@Override

View file

@ -33,7 +33,7 @@
},
"scripts": {
"build": "tsc",
"pretest": "npm run build",
"pretest": "tsc -b tsconfig.bundled.json",
"test": "mocha -r test/helpers.js"
},
"bugs": {
@ -46,8 +46,8 @@
"@types/mocha": "^7.0.2",
"@types/node": "^14.0.4",
"chai": "^4.3.4",
"mocha": "^8.4.0",
"typescript": "^3.8.3"
"mocha": "^10.0.0",
"typescript": "^4.5.4"
},
"dependencies": {
"@keymanapp/models-wordbreakers": "*"

View file

@ -7,8 +7,10 @@ var _ = global;
// TODO: then mocha invocation is as follows:
// TODO: mocha -r @keymanapp/models-test-helpers test/
// KMW string must be included, so do it here:
require('@keymanapp/web-utils');
// Ensure that we can successfully load the module & apply kmwLength, as it's
// needed for some of the unit tests.
require('../build/index.bundled.js');
assert.ok('💩'.kmwLength);
/**
@ -21,7 +23,7 @@ _.jsonFixture = function (name) {
/**
* Returns the Context of an empty buffer; no text, at both the start and
* end of the buffer.
*
*
* @returns {Context}
*/
_.emptyContext = function emptyContext() {

View file

@ -3,7 +3,7 @@
*/
var assert = require('chai').assert;
var models = require('../').models;
var models = require('../build/index.bundled.js').models;
describe('Common utility functions', function() {
// TODO: unit tests for other common utility functions
@ -97,7 +97,7 @@ describe('Common utility functions', function() {
let apple = {
insert: 'apple',
deleteLeft: 0,
deleteRight: 2
deleteRight: 2
};
let banana = {
@ -143,7 +143,7 @@ describe('Common utility functions', function() {
let apple = {
insert: 'apple',
deleteLeft: 0,
deleteRight: 2
deleteRight: 2
};
let banana = {
@ -167,7 +167,7 @@ describe('Common utility functions', function() {
let apple = {
insert: 'apple',
deleteLeft: 0,
deleteRight: 2
deleteRight: 2
};
let banana = {

View file

@ -3,12 +3,12 @@
*/
var assert = require('chai').assert;
var PriorityQueue = require('../').models.PriorityQueue;
var PriorityQueue = require('../build/index.bundled.js').models.PriorityQueue;
describe('Priority queue', function() {
it('can act as a min-heap', function () {
let input = [1, 10, 2, 9, 3, 8, 4, 7, 5, 6];
let queue = new PriorityQueue((a, b) => a - b);
input.forEach((input) => queue.enqueue(input));
@ -41,7 +41,7 @@ describe('Priority queue', function() {
it('can act as a max-heap', function () {
let input = [1, 10, 2, 9, 3, 8, 4, 7, 5, 6];
let queue = new PriorityQueue((a, b) => b - a);
input.forEach((input) => queue.enqueue(input));

View file

@ -3,7 +3,7 @@
*/
var assert = require('chai').assert;
var QuoteBehavior = require('../').models.QuoteBehavior;
var QuoteBehavior = require('../build/index.bundled.js').models.QuoteBehavior;
describe('Quote behaviors', function() {
describe('Script directionality', function() {
@ -144,15 +144,15 @@ describe('Quote behaviors', function() {
quotesForKeepSuggestion: { open: `“`, close: `”`},
insertAfterWord: " "
}
assert.throws(function() {
QuoteBehavior.apply(QuoteBehavior.default, "hello", englishPunctuation, QuoteBehavior.default);
});
assert.throws(function() {
QuoteBehavior.apply(QuoteBehavior.useQuotes, "hello", englishPunctuation, QuoteBehavior.default);
});
assert.throws(function() {
QuoteBehavior.apply(QuoteBehavior.noQuotes, "hello", englishPunctuation, QuoteBehavior.default);
});

View file

@ -3,7 +3,7 @@
*/
var assert = require('chai').assert;
var models = require('../').models;
var models = require('../build/index.bundled.js').models;
var wordBreakers = require('@keymanapp/models-wordbreakers').wordBreakers;
describe('Tokenization functions', function() {
@ -132,7 +132,7 @@ describe('Tokenization functions', function() {
left: '', startOfBuffer: true,
right: '', endOfBuffer: true
};
let tokenization = models.tokenize(wordBreakers.default, context);
let expectedResult = {
@ -140,7 +140,7 @@ describe('Tokenization functions', function() {
right: [],
caretSplitsToken: false
};
assert.deepEqual(tokenization, expectedResult);
});
@ -153,7 +153,7 @@ describe('Tokenization functions', function() {
right: [],
caretSplitsToken: false
};
assert.deepEqual(tokenization, expectedResult);
});
@ -163,7 +163,7 @@ describe('Tokenization functions', function() {
left: ' ', startOfBuffer: true,
right: '', endOfBuffer: true
};
let tokenization = models.tokenize(wordBreakers.default, context);
let expectedResult = {
@ -171,7 +171,7 @@ describe('Tokenization functions', function() {
right: [],
caretSplitsToken: false
};
assert.deepEqual(tokenization, expectedResult);
});
@ -233,7 +233,7 @@ describe('Tokenization functions', function() {
case 'ស្រុកខ្មែរ':
return [srok, shiftSpan(khmer, srok.length)]; // array of the two.
case 'កខ្មែរ':
// I'd admittedly be at least somewhat surprised if a real wordbreaker got this
// I'd admittedly be at least somewhat surprised if a real wordbreaker got this
// and similar situations perfectly right... but at least it gives us what
// we need for a test.
return [k, shiftSpan(khmer, k.length)];
@ -289,6 +289,82 @@ describe('Tokenization functions', function() {
assert.deepEqual(tokenization, expectedResult);
});
let midLetterNonbreaker = (text) => {
let customization = {
rules: [{
match: (context) => {
if(context.propertyMatch(null, ["ALetter"], ["MidLetter"], ["eot"])) {
return true;
} else {
return false;
}
},
breakIfMatch: false
}],
propertyMapping: (char) => {
let hyphens = ['\u002d', '\u2010', '\u058a', '\u30a0'];
if(hyphens.includes(char)) {
return "MidLetter";
} else {
return null;
}
}
};
return wordBreakers.default(text, customization);
}
it('treats caret as `eot` for pre-caret text', function() {
let context = {
left: "don-", // We use a hyphen here b/c single-quote is hardcoded.
right: " worry",
endOfBuffer: true,
startOfBuffer: true
};
let tokenization = models.tokenize(wordBreakers.default, context);
assert.deepEqual(tokenization, {
left: ["don", "-"],
right: ["worry"],
caretSplitsToken: false
});
tokenization = models.tokenize(midLetterNonbreaker, context);
assert.deepEqual(tokenization, {
left: ["don-"],
right: ["worry"],
caretSplitsToken: false
});
});
it('handles mid-contraction tokenization', function() {
let context = {
left: "don:",
right: "t worry",
endOfBuffer: true,
startOfBuffer: true
};
let tokenization = models.tokenize(wordBreakers.default, context);
assert.deepEqual(tokenization, {
left: ["don", ":"], // This particular case feels like a possible issue.
right: ["t", "worry"], // It'd be a three-way split token, as "don:t" would
// be a single token were it not for the caret in the middle.
caretSplitsToken: false
})
tokenization = models.tokenize(midLetterNonbreaker, context);
assert.deepEqual(tokenization, {
left: ["don:"],
right: ["t", "worry"],
caretSplitsToken: true
});
});
});
describe('getLastPreCaretToken', function() {

View file

@ -3,7 +3,7 @@
*/
var assert = require('chai').assert;
var TrieModel = require('../').models.TrieModel;
var TrieModel = require('../build/index.bundled.js').models.TrieModel;
describe('LMLayerWorker trie model for word lists', function() {
describe('instantiation', function () {
@ -20,7 +20,7 @@ describe('LMLayerWorker trie model for word lists', function() {
}
}
})
assert.equal(model.punctuation.insertAfterWord, spaceMark);
assert.equal(model.punctuation.quotesForKeepSuggestion.open, openQuote);
assert.equal(model.punctuation.quotesForKeepSuggestion.close, closeQuote);
@ -176,7 +176,7 @@ describe('LMLayerWorker trie model for word lists', function() {
);
});
});
describe('The default key function', function () {
it('uses the default key function', function () {
var model = new TrieModel(jsonFixture('tries/accented'));

View file

@ -3,7 +3,7 @@
*/
var assert = require('chai').assert;
var TrieModel = require('../').models.TrieModel;
var TrieModel = require('../build/index.bundled.js').models.TrieModel;
// Useful for tests related to strings with supplementary pairs.
var smpForUnicode = function(code){
@ -52,7 +52,7 @@ describe('Trie traversal abstractions', function() {
assert.isEmpty(child.traversal().entries);
for(tChild of traversalInner1.children()) {
if(tChild.char == 'h') {
if(tChild.char == 'h') {
hSuccess = true;
let traversalInner2 = tChild.traversal();
assert.isDefined(traversalInner2);
@ -64,7 +64,7 @@ describe('Trie traversal abstractions', function() {
eSuccess = true;
let traversalInner3 = hChild.traversal();
assert.isDefined(traversalInner3);
assert.isDefined(traversalInner3.entries);
assert.equal(traversalInner3.entries[0], "the");
@ -104,7 +104,7 @@ describe('Trie traversal abstractions', function() {
assert.isEmpty(child.traversal().entries);
for(tChild of traversalInner1.children()) {
if(tChild.char == 'r') {
if(tChild.char == 'r') {
let traversalInner2 = tChild.traversal();
assert.isDefined(traversalInner2);
assert.isArray(tChild.traversal().entries);
@ -185,7 +185,7 @@ describe('Trie traversal abstractions', function() {
assert.isEmpty(child.traversal().entries);
for(aChild of traversalInner1.children()) {
if(aChild.char == smpP) {
if(aChild.char == smpP) {
pSuccess = true;
let traversalInner2 = aChild.traversal();
assert.isDefined(traversalInner2);

View file

@ -0,0 +1,21 @@
{
// This variant of the tsconfig.json exists to create a 'leaf', 'bundled'
// version of the models/templates build product. The same reference
// cannot be prepended twice in a composite tsc build, posing problems
// for certain down-line builds if the two tsconfigs are not differentiated.
"extends": "./tsconfig.json",
"compilerOptions": {
"outFile": "build/index.bundled.js",
},
"references": [
{ "path": "../../web/keyman-version", "prepend": true},
{ "path": "../../web/utils", "prepend": true },
{ "path": "../types" }
],
"include": [
"src/**/*.ts"
],
"exclude": [
"test"
]
}

View file

@ -14,7 +14,7 @@
],
"homepage": "https://github.com/keymanapp/keyman",
"license": "MIT",
"main": "index.js",
"main": "build/index.js",
"directories": {
"lib": "lib",
"test": "test"
@ -43,7 +43,7 @@
"@types/chai": "^4.2.11",
"@types/mocha": "^7.0.2",
"chai": "^4.3.4",
"mocha": "^8.4.0",
"mocha": "^10.0.0",
"ts-node": "^9.1.1",
"typescript": "^4.5.4"
},

View file

@ -4,7 +4,7 @@ export namespace data {
/**
* Valid values for a word break property.
*/
export const enum WordBreakProperty {
export const enum WordBreakProperty { // Scary bit: this does not exist as an object at run-time!
Other,
LF,
Newline,
@ -28,6 +28,32 @@ export const enum WordBreakProperty {
eot
};
// Not currently built by the auto-generator tool, but it easily could be.
// If and when we import the data.ts rebuilder, we can add this in.
export const propertyMap = [
"Other",
"LF",
"Newline",
"CR",
"WSegSpace",
"Double_Quote",
"Single_Quote",
"MidNum",
"MidNumLet",
"Numeric",
"MidLetter",
"ALetter",
"ExtendNumLet",
"Format",
"Extend",
"Hebrew_Letter",
"ZWJ",
"Katakana",
"Regional_Indicator",
"sot",
"eot"
];
/**
* Constants for indexing values in WORD_BREAK_PROPERTY.
*/

View file

@ -1,20 +1,48 @@
// Include the word-breaking data here:
/// <reference path="./data.ts" />
namespace wordBreakers {
/**
* A set of options used to customize and extend the behavior of the default
* Unicode wordbreaker.
*/
export interface DefaultWordBreakerOptions {
/**
* Allows addition of custom wordbreaking rules, which will be applied
* after WB1-WB4 and before all other default wordbreaking rules.
*
* @see `WordbreakerRule`
*/
rules?: WordbreakerRule[];
/**
* Allows assignment of characters to different word-breaking properties than
* their standard word-breaking assignment, including to custom properties
* specified within `customProperties`.
* @param char
*/
propertyMapping?(char: string): string;
/**
* Allows definition of extra word-breaking properties for use with custom
* rules.
*/
customProperties?: string[];
}
/**
* Word breaker based on Unicode Standard Annex #29, Section 4.1:
* Default Word Boundary Specification.
*
* Default Word Boundary Specification.
*
* @see http://unicode.org/reports/tr29/#Word_Boundaries
* @see https://github.com/eddieantonio/unicode-default-word-boundary/tree/v12.0.0
*/
export function default_(text: string): Span[] {
let boundaries = findBoundaries(text);
export function default_(text: string, options?: DefaultWordBreakerOptions): Span[] {
let boundaries = findBoundaries(text, options);
if (boundaries.length == 0) {
return [];
}
// All non-empty strings have at least TWO boundaries at the start and end of
// All non-empty strings have at least TWO boundaries: at the start and at the end of
// the string.
let spans = [];
for (let i = 0; i < boundaries.length - 1; i++) {
@ -22,7 +50,7 @@ namespace wordBreakers {
let end = boundaries[i + 1];
let span = new LazySpan(text, start, end);
if (isNonSpace(span.text)) {
if (isNonSpace(span.text, options)) {
spans.push(span);
// Preserve a sequence-final space if it exists. Needed to signal "end of word".
} else if (i == boundaries.length - 2) { // if "we just checked the final boundary"...
@ -61,13 +89,198 @@ namespace wordBreakers {
}
}
/**
* An abstraction supporting custom wordbreaker boundary rules. While this doesn't provide
* support for more complex rules like WB4, WB15, or WB16, this is sufficient for all other
* default word-breaking rules and can be used to define custom rules of similar structure.
*
* @see https://unicode.org/reports/tr29/#WB_Rule_Macros
*/
export interface WordbreakerRule {
/**
* Indicates whether or not the rule applies in the specified context.
* @param context
*/
match(context: BreakerContext): boolean;
/**
* Indicates whether or not the rule indicates a word boundary at the context's site when it matches.
*/
breakIfMatch: boolean;
}
/**
* Provides a useful presentation for wordbreaker's context for use in word-breaking rules.
*
* @see https://unicode.org/reports/tr29/#Word_Boundary_Rules
*/
export class BreakerContext {
// Referenced by this object in order to facilitate `lookahead` maintenance.
private readonly text: string;
readonly options?: DefaultWordBreakerOptions;
/**
* Represents the property of character immediately preceding `left`'s character.
*/
readonly lookbehind: WordBreakProperty = WordBreakProperty.sot;
/**
* Represents the property of the character immediately preceding the potential word boundary.
*/
readonly left: WordBreakProperty = WordBreakProperty.sot;
/**
* Represents the property of the character immediately following the potential word boundary.
*/
readonly right: WordBreakProperty = WordBreakProperty.sot;
/**
* Represents the property of the character immediately following `right`'s character.
*/
readonly lookahead: WordBreakProperty; // Always initialized by constructor.
/**
* Initializes the word-breaking context at the start of the word-breaker's boundary-detection
* algorithm.
* @param text The text to be word-broken
* @param lookaheadPos The position corresponding to `lookahead`.
*/
constructor(text: string, options: DefaultWordBreakerOptions | undefined, lookaheadPos: number);
/**
* Used internally by the boundary-detection algorithm during context-shifting operations.
* @param text
* @param lookbehind
* @param left
* @param right
* @param lookahead
*/
constructor(text: string,
options: DefaultWordBreakerOptions | undefined,
lookbehind: WordBreakProperty,
left: WordBreakProperty,
right: WordBreakProperty,
lookahead: WordBreakProperty);
constructor(text: string,
options: DefaultWordBreakerOptions | undefined,
prop1: WordBreakProperty | number,
prop2?: WordBreakProperty,
prop3?: WordBreakProperty,
prop4?: WordBreakProperty) {
this.text = text;
this.options = options;
if(arguments.length == 3) {
this.lookahead = this.wordbreakPropertyAt(prop1);// prop1;
} else /*if(arguments.length == 6)*/ {
this.lookbehind = prop1 as WordBreakProperty;
this.left = prop2 as WordBreakProperty;
this.right = prop3 as WordBreakProperty;
this.lookahead = prop4 as WordBreakProperty;
}
}
/**
* The general use-case when shifting boundary-check position if WB4 is not active.
* @param lookahead The WordBreakProperty for the character to become `lookahead`.
* @returns
*/
public next(lookaheadPos: number): BreakerContext {
let newLookahead = this.wordbreakPropertyAt(lookaheadPos);
return new BreakerContext(this.text, this.options, this.left, this.right, this.lookahead, newLookahead);
}
/**
* Used for WB4: when ignoring characters before an intervening linebreak, we
* replace `right` with the current `lookahead`, without affecting `lookbehind`
* or `left`. A new `lookahead` is then needed.
* @param lookahead
* @returns
*/
public ignoringRight(lookaheadPos: number) {
let newLookahead = this.wordbreakPropertyAt(lookaheadPos);
return new BreakerContext(this.text, this.options, this.lookbehind, this.left, this.lookahead, newLookahead);
}
/**
* Used for WB4: when ignoring characters after an intervening linebreak, it's
* `lookahead` that gets replaced without shifting the other tracked properties.
* @param lookahead
* @returns
*/
public ignoringLookahead(lookaheadPos: number) {
let newLookahead = this.wordbreakPropertyAt(lookaheadPos);
return new BreakerContext(this.text, this.options, this.lookbehind, this.left, this.right, newLookahead);
}
/**
* Return the value of the Word_Break property at the given string index.
* @param pos position in the text.
*/
private wordbreakPropertyAt(pos: number) {
if (pos < 0) {
return WordBreakProperty.sot; // Always "start of string" before the string starts!
} else if (pos >= this.text.length) {
return WordBreakProperty.eot; // Always "end of string" after the string ends!
} else if (isStartOfSurrogatePair(this.text[pos])) {
// Surrogate pairs the next TWO items from the string!
return property(this.text[pos] + this.text[pos + 1]);
}
return property(this.text[pos], this.options);
}
/**
* Returns `true` if and only if each member of the context has a property included within
* its corresponding set (when specified). Any set may be replaced with null to disable
* a check against its corresponding property.
* @param lookbehindSet
* @param leftSet
* @param rightSet
* @param lookaheadSet
*/
public match(lookbehindSet: WordBreakProperty[] | null,
leftSet: WordBreakProperty[] | null,
rightSet: WordBreakProperty[] | null,
lookaheadSet: WordBreakProperty[] | null) : boolean {
let result: boolean = lookbehindSet?.includes(this.lookbehind) ?? true;
result = result && (leftSet?.includes(this.left) ?? true);
result = result && (rightSet?.includes(this.right) ?? true);
return result && (lookaheadSet?.includes(this.lookahead) ?? true);
}
/**
* Returns `true` if and only if each member of the context has a property included within
* its corresponding set (when specified). Any set may be replaced with null to disable
* a check against its corresponding property.
*
* Names should match those found at https://unicode.org/reports/tr29/#Word_Boundary_Rules
* or defined in the word-breaker customization options; matching is case-insensitive.
* Also includes two extra properties:
* - `sot` - start of text
* - `eot` - end of text
* @param lookbehindSet
* @param leftSet
* @param rightSet
* @param lookaheadSet
*/
public propertyMatch(lookbehindSet: string[] | null,
leftSet: string[] | null,
rightSet: string[] | null,
lookaheadSet: string[] | null) : boolean {
const propMapper = (name: string) => propertyVal(name, this.options);
return this.match(lookbehindSet?.map(propMapper) as WordBreakProperty[] | null,
leftSet?.map(propMapper) as WordBreakProperty[] | null,
rightSet?.map(propMapper) as WordBreakProperty[] | null,
lookaheadSet?.map(propMapper) as WordBreakProperty[] | null);
}
}
/**
* Returns true when the chunk does not solely consist of whitespace.
*
* @param chunk a chunk of text. Starts and ends at word boundaries.
*/
function isNonSpace(chunk: string): boolean {
return !Array.from(chunk).map(property).every(wb => (
function isNonSpace(chunk: string, options?: DefaultWordBreakerOptions): boolean {
return !Array.from(chunk).map((char) => property(char, options)).every(wb => (
wb === WordBreakProperty.CR ||
wb === WordBreakProperty.LF ||
wb === WordBreakProperty.Newline ||
@ -82,13 +295,17 @@ namespace wordBreakers {
*
* @param text Text to find word boundaries in.
*/
function findBoundaries(text: string): number[] {
function findBoundaries(text: string, options?: DefaultWordBreakerOptions): number[] {
// WB1 and WB2: no boundaries if given an empty string.
if (text.length === 0) {
// There are no boundaries in an empty string!
return [];
}
if(options && !options.rules) {
options.rules = [];
}
// This algorithm works by maintaining a sliding window of four SCALAR VALUES.
//
// - Scalar values? JavaScript strings are NOT actually a string of
@ -112,10 +329,7 @@ namespace wordBreakers {
let rightPos: number;
let lookaheadPos = 0; // lookahead, one scalar value to the right of right.
// Before the start of the string is also the start of the string.
let lookbehind: WordBreakProperty;
let left = WordBreakProperty.sot;
let right = WordBreakProperty.sot;
let lookahead = wordbreakPropertyAt(0);
let state = new BreakerContext(text, options, lookaheadPos);
// Count RIs to make sure we're not splitting emoji flags:
let nConsecutiveRegionalIndicators = 0;
@ -124,34 +338,32 @@ namespace wordBreakers {
rightPos = lookaheadPos;
lookaheadPos = positionAfter(lookaheadPos);
// Shift all properties, one scalar value to the right.
[lookbehind, left, right, lookahead] =
[left, right, lookahead, wordbreakPropertyAt(lookaheadPos)];
state = state.next(lookaheadPos);
// Break at the start and end of text, unless the text is empty.
// WB1: Break at start of text...
if (left === WordBreakProperty.sot) {
if (state.match(null, [WordBreakProperty.sot], null, null)) {
boundaries.push(rightPos);
continue;
}
// WB2: Break at the end of text...
if (right === WordBreakProperty.eot) {
if (state.match(null, null, [WordBreakProperty.eot], null)) {
boundaries.push(rightPos);
break; // Reached the end of the string. We're done!
}
// WB3: Do not break within CRLF:
if (left === WordBreakProperty.CR && right === WordBreakProperty.LF)
if (state.match(null, [WordBreakProperty.CR], [WordBreakProperty.LF], null)) {
continue;
}
// WB3b: Otherwise, break after...
if (left === WordBreakProperty.Newline ||
left === WordBreakProperty.CR ||
left === WordBreakProperty.LF) {
const NEWLINE_SET = [WordBreakProperty.Newline, WordBreakProperty.CR, WordBreakProperty.LF];
if(state.match(null, NEWLINE_SET, null, null)) {
boundaries.push(rightPos);
continue;
}
// WB3a: ...and before newlines
if (right === WordBreakProperty.Newline ||
right === WordBreakProperty.CR ||
right === WordBreakProperty.LF) {
if (state.match(null, null, NEWLINE_SET, null)) {
boundaries.push(rightPos);
continue;
}
@ -163,107 +375,144 @@ namespace wordBreakers {
// https://www.unicode.org/Public/emoji/12.0/emoji-zwj-sequences.txt
// WB3d: Keep horizontal whitespace together
if (left === WordBreakProperty.WSegSpace && right == WordBreakProperty.WSegSpace)
if (state.match(null, [WordBreakProperty.WSegSpace], [WordBreakProperty.WSegSpace], null)) {
continue;
}
// WB4: Ignore format and extend characters
// This is to keep grapheme clusters together!
// See: Section 6.2: https://unicode.org/reports/tr29/#Grapheme_Cluster_and_Format_Rules
// N.B.: The rule about "except after sot, CR, LF, and
// Newline" already been by WB1, WB2, WB3a, and WB3b above.
while (right === WordBreakProperty.Format ||
right === WordBreakProperty.Extend ||
right === WordBreakProperty.ZWJ) {
const SET_WB4_IGNORE = [WordBreakProperty.Format, WordBreakProperty.Extend, WordBreakProperty.ZWJ];
while (state.match(null, null, SET_WB4_IGNORE, null)) {
// Continue advancing in the string, as if these
// characters do not exist. DO NOT update left and
// lookbehind however!
[rightPos, lookaheadPos] = [lookaheadPos, positionAfter(lookaheadPos)];
[right, lookahead] = [lookahead, wordbreakPropertyAt(lookaheadPos)];
state = state.ignoringRight(lookaheadPos);
}
// In ignoring the characters in the previous loop, we could
// have fallen off the end of the string, so end the loop
// prematurely if that happens!
if (right === WordBreakProperty.eot) {
if (state.right === WordBreakProperty.eot) {
boundaries.push(rightPos);
break;
}
// WB4 (continued): Lookahead must ALSO ignore these format,
// extend, ZWJ characters!
while (lookahead === WordBreakProperty.Format ||
lookahead === WordBreakProperty.Extend ||
lookahead === WordBreakProperty.ZWJ) {
while (state.match(null, null, null, SET_WB4_IGNORE)) {
// Continue advancing in the string, as if these
// characters do not exist. DO NOT update left and right,
// however!
lookaheadPos = positionAfter(lookaheadPos);
lookahead = wordbreakPropertyAt(lookaheadPos);
state = state.ignoringLookahead(lookaheadPos);
}
// See: https://unicode.org/reports/tr29/#WB_Rule_Macros
const SET_AHLETTER = [WordBreakProperty.ALetter, WordBreakProperty.Hebrew_Letter];
const SET_MIDNUMLETQ = [WordBreakProperty.MidNumLet, WordBreakProperty.Single_Quote];
// Custom rules may override the base ruleset aside from the first few fundamental ones.
if(options?.rules) {
let customMatch: boolean = false;
for(const rule of options.rules) {
customMatch = rule.match(state);
if(customMatch) {
if(rule.breakIfMatch) {
boundaries.push(rightPos);
}
break; // as customMatch == true here, this will trigger the `continue` that follows.
}
}
if(customMatch) {
continue;
}
}
// WB5: Do not break between most letters.
if (isAHLetter(left) && isAHLetter(right))
// if (isAHLetter(state.left) && isAHLetter(state.right))
if(state.match(null, SET_AHLETTER, SET_AHLETTER, null)) {
continue;
}
// Do not break across certain punctuation
// WB6: (Don't break before apostrophes in contractions)
if (isAHLetter(left) && isAHLetter(lookahead) &&
(right === WordBreakProperty.MidLetter || isMidNumLetQ(right)))
const SET_ALL_MIDLETTER = [WordBreakProperty.MidLetter, ...SET_MIDNUMLETQ];
if(state.match(null, SET_AHLETTER, SET_ALL_MIDLETTER, SET_AHLETTER)) {
continue;
}
// WB7: (Don't break after apostrophes in contractions)
if (isAHLetter(lookbehind) && isAHLetter(right) &&
(left === WordBreakProperty.MidLetter || isMidNumLetQ(left)))
if(state.match(SET_AHLETTER, SET_ALL_MIDLETTER, SET_AHLETTER, null)) {
continue;
}
// WB7a
if (left === WordBreakProperty.Hebrew_Letter && right === WordBreakProperty.Single_Quote)
if(state.match(null, [WordBreakProperty.Hebrew_Letter], [WordBreakProperty.Single_Quote], null)) {
continue;
}
// WB7b
if (left === WordBreakProperty.Hebrew_Letter && right === WordBreakProperty.Double_Quote &&
lookahead === WordBreakProperty.Hebrew_Letter)
if(state.match(null,
[WordBreakProperty.Hebrew_Letter],
[WordBreakProperty.Double_Quote],
[WordBreakProperty.Hebrew_Letter])) {
continue;
}
// WB7c
if (lookbehind === WordBreakProperty.Hebrew_Letter && left === WordBreakProperty.Double_Quote &&
right === WordBreakProperty.Hebrew_Letter)
if(state.match([WordBreakProperty.Hebrew_Letter],
[WordBreakProperty.Double_Quote],
[WordBreakProperty.Hebrew_Letter],
null)) {
continue;
}
// Do not break within sequences of digits, or digits adjacent to letters.
// e.g., "3a" or "A3"
// WB8
if (left === WordBreakProperty.Numeric && right === WordBreakProperty.Numeric)
if(state.match(null, [WordBreakProperty.Numeric], [WordBreakProperty.Numeric], null)) {
continue;
}
// WB9
if (isAHLetter(left) && right === WordBreakProperty.Numeric)
if(state.match(null, SET_AHLETTER, [WordBreakProperty.Numeric], null)) {
continue;
}
// WB10
if (left === WordBreakProperty.Numeric && isAHLetter(right))
if(state.match(null, [WordBreakProperty.Numeric], SET_AHLETTER, null)) {
continue;
}
// Do not break within sequences, such as 3.2, 3,456.789
// WB11
if (lookbehind === WordBreakProperty.Numeric && right === WordBreakProperty.Numeric &&
(left === WordBreakProperty.MidNum || isMidNumLetQ(left)))
const SET_ALL_MIDNUM = [WordBreakProperty.MidNum, ...SET_MIDNUMLETQ];
if(state.match([WordBreakProperty.Numeric], SET_ALL_MIDNUM, [WordBreakProperty.Numeric], null)) {
continue;
}
// WB12
if (left === WordBreakProperty.Numeric && lookahead === WordBreakProperty.Numeric &&
(right === WordBreakProperty.MidNum || isMidNumLetQ(right)))
if(state.match(null, [WordBreakProperty.Numeric], SET_ALL_MIDNUM, [WordBreakProperty.Numeric])) {
continue;
}
// WB13: Do not break between Katakana
if (left === WordBreakProperty.Katakana && right === WordBreakProperty.Katakana)
if(state.match(null, [WordBreakProperty.Katakana], [WordBreakProperty.Katakana], null)) {
continue;
}
// Do not break from extenders (e.g., U+202F NARROW NO-BREAK SPACE)
// WB13a
if ((isAHLetter(left) ||
left === WordBreakProperty.Numeric ||
left === WordBreakProperty.Katakana ||
left === WordBreakProperty.ExtendNumLet) &&
right === WordBreakProperty.ExtendNumLet)
const SET_NUM_KAT_LET = [WordBreakProperty.Katakana,
WordBreakProperty.Numeric,
...SET_AHLETTER];
if(state.match(null, SET_NUM_KAT_LET, [WordBreakProperty.ExtendNumLet], null)) {
continue;
}
if(state.match(null, [WordBreakProperty.ExtendNumLet], [WordBreakProperty.ExtendNumLet], null)) {
continue;
}
// WB13b
if ((isAHLetter(right) ||
right === WordBreakProperty.Numeric ||
right === WordBreakProperty.Katakana) && left === WordBreakProperty.ExtendNumLet)
if(state.match(null, [WordBreakProperty.ExtendNumLet], SET_NUM_KAT_LET, null)) {
continue;
}
// WB15 & WB16:
// Do not break within emoji flag sequences. That is, do not break between
// regional indicator (RI) symbols if there is an odd number of RI
// characters before the break point.
if (right === WordBreakProperty.Regional_Indicator) {
if (state.right === WordBreakProperty.Regional_Indicator) {
// Emoji flags are actually composed of TWO scalar values, each being a
// "regional indicator". These indicators correspond to Latin letters. Put
// two of them together, and they spell out an ISO 3166-1-alpha-2 country
@ -300,34 +549,6 @@ namespace wordBreakers {
}
return pos + 1;
}
/**
* Return the value of the Word_Break property at the given string index.
* @param pos position in the text.
*/
function wordbreakPropertyAt(pos: number) {
if (pos < 0) {
return WordBreakProperty.sot; // Always "start of string" before the string starts!
} else if (pos >= text.length) {
return WordBreakProperty.eot; // Always "end of string" after the string ends!
} else if (isStartOfSurrogatePair(text[pos])) {
// Surrogate pairs the next TWO items from the string!
return property(text[pos] + text[pos + 1]);
}
return property(text[pos]);
}
// Word_Break rule macros
// See: https://unicode.org/reports/tr29/#WB_Rule_Macros
function isAHLetter(prop: WordBreakProperty): boolean {
return prop === WordBreakProperty.ALetter ||
prop === WordBreakProperty.Hebrew_Letter;
}
function isMidNumLetQ(prop: WordBreakProperty): boolean {
return prop === WordBreakProperty.MidNumLet ||
prop === WordBreakProperty.Single_Quote;
}
}
function isStartOfSurrogatePair(character: string) {
@ -340,15 +561,37 @@ namespace wordBreakers {
* Note that
* @param character a scalar value
*/
function property(character: string): WordBreakProperty {
function property(character: string, options?: DefaultWordBreakerOptions): WordBreakProperty {
// If there is a customized mapping for the character, prioritize that.
if(options?.propertyMapping) {
let propName = options.propertyMapping(character);
if(propName) {
return propertyVal(propName, options);
}
}
// This MUST be a scalar value.
// TODO: remove dependence on character.codepointAt()?
let codepoint = character.codePointAt(0) as number;
return searchForProperty(codepoint, 0, WORD_BREAK_PROPERTY.length - 1);
}
function propertyVal(propName: string, options?: DefaultWordBreakerOptions) {
const matcher = (name: string) => name.toLowerCase() == propName.toLowerCase()
const customIndex = options?.customProperties?.findIndex(matcher) ?? -1;
return customIndex != -1 ? -customIndex - 1 : data.propertyMap.findIndex(matcher);
}
/**
* Binary search for the word break property of a given CODE POINT.
*
* The auto-generated data.ts master array defines a **character range**
* lookup table. If a character's codepoint is equal to or greater than
* the I.Start value for an entry and exclusively less than the next entry,
* it falls in the first entry's range bucket and is classified accordingly
* by this method.
*/
function searchForProperty(codePoint: number, left: number, right: number): WordBreakProperty {
// All items that are not found in the array are assigned the 'Other' property.

View file

@ -1,5 +1,5 @@
var assert = require('chai').assert;
var breakASCIIWords = require('..').wordBreakers['ascii'];
var breakASCIIWords = require('../build').wordBreakers['ascii'];
describe('The ASCII word breaker', function () {
it('should break simple English sentences', function () {

View file

@ -3,21 +3,339 @@
*/
const assert = require('chai').assert;
const breakWords = require('..').wordBreakers['default'];
const breakWords = require('../build').wordBreakers['default'];
const SHY = '\u00AD';
const SHY = '\u00AD'; // Other, Format. The "Soft HYphen" - usually invisible unless needed for word-wrapping.
describe('The default word breaker', function () {
it('should break multilingual text', function () {
let breaks = breakWords(
`Добрый день! ᑕᐻ᙮ — after working on ka${SHY}wen${SHY}non:${SHY}nis,
let's eat phở! 🥣`
);
let words = breaks.map(span => span.text);
assert.deepEqual(words, [
'Добрый', 'день', '!', 'ᑕᐻ', '᙮', '—', 'after',
'working', 'on', `ka${SHY}wen${SHY}non:${SHY}nis`, ',',
"let's", 'eat', 'phở', '!', '🥣'
]);
describe('default configuration', function() {
it('should break multilingual text', function () {
let breaks = breakWords(
`Добрый день! ᑕᐻ᙮ — after working on ka${SHY}wen${SHY}non:${SHY}nis,
let's eat phở! 🥣`
);
let words = breaks.map(span => span.text);
assert.deepEqual(words, [
'Добрый', 'день', '!', 'ᑕᐻ', '᙮', '—', 'after',
'working', 'on', `ka${SHY}wen${SHY}non:${SHY}nis`, ',',
"let's", 'eat', 'phở', '!', '🥣'
]);
});
it('handles heavily-punctuated English text', function() {
// This test case brought to you by http://unicode.org/reports/tr29/#Word_Boundaries, Figure 1.
let breaks = breakWords(
`The quick ("brown") fox can't jump 32.3 feet, right?`
);
let words = breaks.map(span => span.text);
assert.deepEqual(words, [
'The', 'quick', '(', '"', 'brown', '"', ')', 'fox', "can't",
'jump', '32.3', 'feet', ',', 'right', '?'
]);
});
// The way these two tests are written is a bit much on the "white-box" style,
// but they do decently cover the boundary rules mentioned.
it('Does not split empty contexts (WB1 + WB2)', function() {
let breaks = breakWords('');
let words = breaks.map(span => span.text);
assert.deepEqual(words, []);
});
it('Does split at context boundaries (WB1 + WB2)', function() {
let breaks = breakWords('a');
let words = breaks.map(span => span.text);
assert.deepEqual(words, ['a']);
});
// WB3, WB3a, WB3b are all handled internally, within the top-level function.
// iff, as in "if and only if"
it('ignores the zero-width joiner iff appropriate (WB4)', function() {
const zwj = '\u200d';
let breaks = breakWords(`a${zwj}b\n${zwj}c${zwj}\nd`);
let words = breaks.map(span => span.text);
// Does NOT ignore the zwj immediately after a newline - the notable exception
// (the reason for "iff", not "if").
assert.deepEqual(words, [`a${zwj}b`, `${zwj}`, `c${zwj}`, `d`]);
})
it('ignores extend characters iff appropriate (WB4)', function() {
const comboGrave = '\u0300'; // The 'combining grave accent', as used in NFD.
let breaks = breakWords(`a${comboGrave}e\n${comboGrave}i${comboGrave}\no`);
let words = breaks.map(span => span.text);
// Does NOT ignore the zwj immediately after a newline - the notable exception
// (the reason for "iff", not "if").
assert.deepEqual(words, [`a${comboGrave}e`, `${comboGrave}`, `i${comboGrave}`, `o`]);
});
it('ignores format characters iff appropriate (WB4)', function() {
// Re-uses `const SHY` from above.
let breaks = breakWords(`a${SHY}e\n${SHY}i${SHY}\no`);
let words = breaks.map(span => span.text);
// Does NOT ignore the zwj immediately after a newline - the notable exception
// (the reason for "iff", not "if").
assert.deepEqual(words, [`a${SHY}e`, `${SHY}`, `i${SHY}`, `o`]);
});
it('does not break between most alphabetic characters (WB5)', function() {
let breaks = breakWords(`aəσאБ лאʈγX`); // a mix of latin, hebrew, greek, cyrillic, and IPA chars
// for both "words".
let words = breaks.map(span => span.text);
assert.deepEqual(words, [`aəσאБ`, `лאʈγX`]);
});
it('does not break letters across specific punctuation patterns (WB6, WB7)', function() {
// `'`: MidNumLetQ (from Single_Quote)
// '.': MidNumLet
// ':': MidLetter
let breaks = breakWords(`don't b.r.e.a.k t:h:e:s:e`);
let words = breaks.map(span => span.text);
assert.deepEqual(words, [`don't`, `b.r.e.a.k`, `t:h:e:s:e`]);
let breaks2 = breakWords(`.drop: :the' 'extras.`);
let words2 = breaks2.map(span => span.text);
assert.deepEqual(words2, [`.`, `drop`, `:`, `:`, `the`, `'`, `'`, `extras`, `.`]);
// ',': MidNum (is NOT included by rule!)
let breaks3 = breakWords('do br,eak that');
let words3 = breaks3.map(span => span.text);
assert.deepEqual(words3, ['do', 'br', ',', 'eak', 'that']);
});
it('treats Hebrew properly (WB7a-c)', function() {
const aleph = 'א';
const bet = 'ב';
// As Hebrew is RTL... this is probably the clearest way for us LTR people to
// clearly see what's going on without ordering mechanics messing up the render.
let breaks = breakWords(`${aleph}' ${aleph}" ${aleph}"${bet}`);
let words = breaks.map(span => span.text);
// A lingering double-quote isn't cool, but one in the middle's fine.
// Lingering single-quote is fine regardless.
assert.deepEqual(words, [`${aleph}'`, `${aleph}`, `"`, `${aleph}"${bet}`]);
});
it(`doesn't break within digit + digit/letter sequences (WB8-10)`, function() {
let breaks = breakWords('a1b2c3 hunter2 ab12cd34 1234567890');
let words = breaks.map(span => span.text);
assert.deepEqual(words, ['a1b2c3', 'hunter2', 'ab12cd34', '1234567890']);
});
it('does not break within formatted number sequences (WB11-12)', function() {
// Note: `'` fits "MidNumLetQ", part of the two rules!
let breaks = breakWords(`1.2.3 3,458.01 3.45'8,01`);
let words = breaks.map(span => span.text);
assert.deepEqual(words, [`1.2.3`, `3,458.01`, `3.45'8,01`]);
let breaks2 = breakWords(`.1' ,3.`);
let words2 = breaks2.map(span => span.text);
assert.deepEqual(words2, [`.`, `1`, `'`, `,`, `3`, `.`]);
});
it('does not break between Katakana (WB13)', function() {
const kataSmA = '\u30a2'; //ァ
const kataA = '\u30a2'; //ア
const kataSound = '\u309b'; // ゛
let breaks = breakWords(`${kataSound}${kataA} ${kataSmA}${kataSound}b ${kataA}${kataSound}${kataSmA}`);
let words = breaks.map(span => span.text);
assert.deepEqual(words, [
`${kataSound}${kataA}`,
`${kataSmA}${kataSound}`,
'b',
`${kataA}${kataSound}${kataSmA}`
]);
});
it('does not break form extenders (WB13a-b)', function() {
// The `_` (underscore) fits the ExtendNumLet class this rule focuses on.
const kataA = '\u30a2'; //ア
let breaks = breakWords(`${kataA}_a__0_b_${kataA} _${kataA} 1_ _c_ ____`);
let words = breaks.map(span => span.text);
assert.deepEqual(words, [
`${kataA}_a__0_b_${kataA}`,
`_${kataA}`,
`1_`,
`_c_`,
`____`
]);
});
it('handles emoji flag sequences properly (WB15-16)', function() {
// For clarity on what's being tested...
let CA_FLAG = '\u{1f1e8}\u{1f1e6}' // '🇨🇦' (canadian flag emoji); should not be broken.
let KH_FLAG = '\u{1f1f0}\u{1f1ed}' // '🇰🇭' (khmer flag emoji); same
let X_FLAG_PIECE = '\u{1f1fd}' // '🇽' (half of a flag emoji; '🇽🇽' doesn't match a flag)
let breaks = breakWords(`${CA_FLAG}${KH_FLAG}${X_FLAG_PIECE}${X_FLAG_PIECE}`);
let words = breaks.map(span => span.text);
// Note that the emoji may not render well within VSCode, but they show up nicely on GitHub.
assert.deepEqual(words, ['🇨🇦', '🇰🇭', '🇽🇽']);
});
it('breaks hyphenated words by default', function() {
let breaks = breakWords('Smith-Jones');
let words = breaks.map(span => span.text);
assert.deepEqual(words, ['Smith', '-', 'Jones']);
});
});
describe('customization', function() {
// Refer to https://unicode.org/reports/tr29/#Word_Boundary_Rules, third bullet point.
it('custom prop, rule: do not break on letter-adjacent hyphens', function() {
let customization = {
rules: [{
match: (context) => {
if(context.propertyMatch(null, ["ALetter"], ["Hyphen"], ["ALetter"])) {
return true;
} else if(context.propertyMatch(["ALetter"], ["Hyphen"], ["ALetter"], null)) {
return true;
} else {
return false;
}
},
breakIfMatch: false
}],
propertyMapping: (char) => {
const validHyphenCodes = [
'\u002d', '\u2010', '\u058a', '\u30a0'
];
if(validHyphenCodes.includes(char)) {
return "Hyphen";
}
return null;
},
customProperties: ["Hyphen"]
}
let breaks = breakWords('Smith-Jones', customization);
let words = breaks.map(span => span.text);
assert.deepEqual(words, ['Smith-Jones']);
});
it('mid-word hyphen via reassignment to MidLetter', function() {
let customization = {
propertyMapping: (char) => {
const validHyphenCodes = [
'\u002d', '\u2010', '\u058a', '\u30a0'
];
if(validHyphenCodes.includes(char)) {
return "MidLetter";
}
return null;
}
}
let breaks = breakWords('Smith-Jones', customization);
let words = breaks.map(span => span.text);
assert.deepEqual(words, ['Smith-Jones']);
});
// Useful for some regional minority languages that prefer word-breaking spaces.
it('character reassignment: Khmer letters as ALetter', function() {
let customization = {
propertyMapping: (char) => {
if(char >= '\u1780' && char <= '\u17b3') {
return "ALetter";
} else {
// The other Khmer characters already have useful word-breaking
// property assignments.
return null;
}
}
}
let breaks = breakWords('ស្រុក ខ្មែរ', customization);
let words = breaks.map(span => span.text);
assert.deepEqual(words, ['ស្រុក', 'ខ្មែរ']);
});
// See: suggested language-specific WB5a from the spec's notes.
it("french/italian apostrophe / vowel boundaries", function() {
let customization = {
rules: [
// WB5, but with differentiated consonants (ALetter) and vowels (AVowel)
{
match: (context) => {
if(context.propertyMatch(null, ["ALetter", "AVowel"], ["ALetter", "AVowel"], null)) {
return true;
} else {
return false;
}
},
breakIfMatch: false
},
// Proposed WB5a
{
match: (context) => {
if(context.propertyMatch(null, ["Single_Quote"], ["AVowel"], null)) {
return true;
} else {
return false;
}
},
breakIfMatch: true
},
// WB6, 7
{
match: (context) => {
if(context.propertyMatch(null,
["ALetter", "AVowel"],
["MidLetter", "MidNumLet", "Single_Quote"],
["ALetter", "AVowel"])) {
return true;
} else if(context.propertyMatch(["ALetter", "AVowel"],
["MidLetter", "MidNumLet", "Single_Quote"],
["ALetter", "AVowel"],
null)) {
return true;
} else {
return false;
}
},
breakIfMatch: false
}
// Similar extensions to WB9, 10, 13a, and 13b would also be needed for robustness.
// And I kind of left the Hebrew_Letter out of the WB5, 6, and 7 rewrites.
],
propertyMapping: (char) => {
const vowels = ['a', 'e', 'i', 'o', 'u'];
if(vowels.includes(char)) {
return "AVowel";
}
return null;
},
customProperties: ["AVowel"]
}
let breaks = breakWords("l'objectif aujourd'hui", customization);
let words = breaks.map(span => span.text);
assert.deepEqual(words, ["l'", "objectif", "aujourd'hui"]);
});
});
});

View file

@ -1,5 +1,5 @@
const assert = require('chai').assert;
const breakWords = require('..').wordBreakers['placeholder'];
const breakWords = require('../build').wordBreakers['placeholder'];
describe('The placeholder word breaker', function () {

View file

@ -37,7 +37,7 @@
"karma-mocha-reporter": "^2.2.5",
"karma-safari-launcher": "^1.0.0",
"karma-teamcity-reporter": "^1.1.0",
"mocha": "^8.4.0",
"mocha": "^10.0.0",
"mocha-teamcity-reporter": "^4.0.0",
"sinon": "^14.0.0",
"ts-node": "^9.1.1",

View file

@ -45,7 +45,7 @@ describe('ContextTracker', function() {
assert.deepEqual(state.tokens.map(token => token.raw), rawTokens);
});
it("properly matches and aligns when a 'wordbreak' is added'", function() {
it("properly matches and aligns when a 'wordbreak' is added", function() {
let existingContext = ["an", "apple", "a", "day", "keeps", "the", "doctor"];
let transform = {
insert: ' ',
@ -56,7 +56,7 @@ describe('ContextTracker', function() {
let rawTokens = ["an", null, "apple", null, "a", null, "day", null, "keeps", null, "the", null, "doctor", null, ""];
let existingState = ContextTracker.modelContextState(existingContext);
let state = ContextTracker.attemptMatchContext(newContext, existingState, null, toWrapperDistribution(transform));
let state = ContextTracker.attemptMatchContext(newContext, existingState, toWrapperDistribution(transform));
assert.isNotNull(state);
assert.deepEqual(state.tokens.map(token => token.raw), rawTokens);
@ -65,6 +65,26 @@ describe('ContextTracker', function() {
assert.isEmpty(state.tokens[state.tokens.length - 1].transformDistributions);
});
it("properly matches and aligns when an implied 'wordbreak' occurs (as when following \"'\")", function() {
let existingContext = ["'"];
let transform = {
insert: 'a',
deleteLeft: 0
}
let newContext = Array.from(existingContext);
newContext.push('a'); // The incoming transform should produce a new token WITH TEXT.
let rawTokens = ["'", null, "a"];
let existingState = ContextTracker.modelContextState(existingContext);
let state = ContextTracker.attemptMatchContext(newContext, existingState, toWrapperDistribution(transform));
assert.isNotNull(state);
assert.deepEqual(state.tokens.map(token => token.raw), rawTokens);
// The 'wordbreak' transform
assert.isEmpty(state.tokens[state.tokens.length - 2].transformDistributions);
assert.isNotEmpty(state.tokens[state.tokens.length - 1].transformDistributions);
});
it("properly matches and aligns when lead token is removed AND a 'wordbreak' is added'", function() {
let existingContext = ["an", "apple", "a", "day", "keeps", "the", "doctor"];
let transform = {
@ -77,7 +97,7 @@ describe('ContextTracker', function() {
let rawTokens = ["apple", null, "a", null, "day", null, "keeps", null, "the", null, "doctor", null, ""];
let existingState = ContextTracker.modelContextState(existingContext);
let state = ContextTracker.attemptMatchContext(newContext, existingState, null, toWrapperDistribution(transform));
let state = ContextTracker.attemptMatchContext(newContext, existingState, toWrapperDistribution(transform));
assert.isNotNull(state);
assert.deepEqual(state.tokens.map(token => token.raw), rawTokens);

View file

@ -0,0 +1,58 @@
var assert = require('chai').assert;
let TransformUtils = require('../../../web/lm-worker/build/intermediate.js').TransformUtils;
describe('TransformUtils', function () {
describe('isWhitespace', function () {
it("should not match a string containing standard alphabetic characters", function () {
let testTransforms = [{
insert: "a ",
deleteLeft: 0
}, {
insert: " a",
deleteLeft: 0
}, {
insert: "ab",
deleteLeft: 0
}];
testTransforms.forEach((transform) => assert.isFalse(TransformUtils.isWhitespace(transform), `failed with: '${transform.insert}'`));
});
it("should match a simple ' ' transform", function() {
transform = {
insert: " ",
deleteLeft: 0
};
assert.isTrue(TransformUtils.isWhitespace(transform));
});
it("should match a simple ' ' transform with delete-left", function() {
transform = {
insert: " ",
deleteLeft: 1
};
assert.isTrue(TransformUtils.isWhitespace(transform));
});
it("should match a transform consisting of multiple characters of only whitespace", function() {
transform = {
insert: " \n\r\u00a0\t\u2000 ",
deleteLeft: 0
};
assert.isTrue(TransformUtils.isWhitespace(transform));
});
it("stress tests", function() {
transform = {
insert: " \n\r\u00a0\ta\u2000 ", // the 'a' should cause failure.
deleteLeft: 0
};
assert.isFalse(TransformUtils.isWhitespace(transform));
});
});
});

View file

@ -50,6 +50,45 @@ describe('ModelCompositor', function() {
});
});
it('strongly avoids corrections for single-character roots', function() {
let compositor = new ModelCompositor(plainModel);
let context = {
left: '', startOfBuffer: true, endOfBuffer: true,
};
// The 'weights' involved imply that we have an edge-case fat finger on the bottom of
// the 'q' key, slightly in its favor.
let inputDistribution = [
{sample: {insert: 'q', deleteLeft: 0}, p: 0.5}, // 'quite' (679) and 'question' (644) are included!
{sample: {insert: 'a', deleteLeft: 0}, p: 0.4} // but at lower weight than 'and' (998).
];
compositor.predict({insert: '', deleteLeft: 0}, context); // Initialize context tracking first!
let suggestions = compositor.predict(inputDistribution, context);
// remove the keep suggestion; we're not testing that here.
suggestions = suggestions.filter((suggestion) => suggestion.tag != 'keep');
suggestions.sort((a, b) => b.p - a.p);
// There are only 4 suggestions in this limited test model that begin with 'q'.
// We expect more than that, since 'a' is indicated to be very close by.
assert.isAbove(suggestions.length, 4, "fat-finger style corrections needed for test comparisons are missing");
// Note: 'and' is (currently) modeled by the text-fixture model to have 9.3x the base probability
// that the worst 'q'-rooted suggestion ('quality') does. Without single-character correction
// avoidance logic, this test _will_ fail.
//
// In case a tweak to test parameters is desired, note that 'and' beats rank #3 - 'questions' -
// at 3.36x base. At the time of writing this test, upping 'a's probability to 0.45 will block
// 'quality' while the top three 'q's (ending with 'questions') remain in place.
let qRange = suggestions.slice(0, 4);
assert.isUndefined(qRange.find((suggestion) => suggestion.transform.insert.charAt(0) != 'q'));
let aRange = suggestions.slice(4);
assert.isUndefined(aRange.find((suggestion) => suggestion.transform.insert.charAt(0) == 'q'));
});
it('properly handles suggestions after a backspace', function() {
let compositor = new ModelCompositor(plainModel);
let context = {
@ -66,12 +105,53 @@ describe('ModelCompositor', function() {
// Suggestions always delete the full root of the suggestion.
//
// After a backspace, that means the text 'the' - 3 chars.
// Char 4 is for the original backspace, as suggstions are built
// Char 4 is for the original backspace, as suggestions are built
// based on the context state BEFORE the triggering input -
// here, a backspace.
assert.equal(suggestion.transform.deleteLeft, 4);
});
});
it('properly handles suggestions for the first letter after a ` `', function() {
let compositor = new ModelCompositor(plainModel);
let context = {
left: 'the', startOfBuffer: true, endOfBuffer: true,
};
let inputTransform = {
insert: ' ',
deleteLeft: 0
};
let suggestions = compositor.predict(inputTransform, context);
suggestions.forEach(function(suggestion) {
// After a space, predictions are based on a new, zero-length root.
// With nothing to replace, .deleteLeft should be zero.
assert.equal(suggestion.transform.deleteLeft, 0);
});
});
it('properly handles suggestions for the first letter after a `\'`', function() {
let compositor = new ModelCompositor(plainModel);
let context = {
left: "the '", startOfBuffer: true, endOfBuffer: true,
};
// This results in a new word boundary (between the `'` and the `a`).
// Basically, an implied (but nonexistent) ` `.
let inputTransform = {
insert: "a",
deleteLeft: 0
};
let suggestions = compositor.predict(inputTransform, context);
suggestions.forEach(function(suggestion) {
// Suggestions always delete the full root of the suggestion.
// Which, here, didn't exist before the input. Nothing to
// replace => nothing for the suggestion to delete.
assert.equal(suggestion.transform.deleteLeft, 0);
});
});
});
describe('applySuggestionCasing', function() {

View file

@ -5,6 +5,7 @@ THIS_SCRIPT="$(greadlink -f "${BASH_SOURCE[0]}" 2>/dev/null || readlink -f "${BA
. "$(dirname "$THIS_SCRIPT")/../../../resources/build/build-utils.sh"
## END STANDARD BUILD SCRIPT INCLUDE
. "$KEYMAN_ROOT/resources/build/build-utils-ci.inc.sh"
. "$KEYMAN_ROOT/resources/shellHelperFunctions.sh"
SCRIPT_ROOT="$(dirname "$THIS_SCRIPT")"
@ -33,6 +34,18 @@ test-headless ( ) {
_FLAGS="$_FLAGS --reporter mocha-teamcity-reporter"
fi
pushd "$KEYMAN_ROOT/common/models/wordbreakers"
npm run test || fail "models/wordbreakers tests failed"
popd
pushd "$KEYMAN_ROOT/common/models/templates"
npm run test || fail "models/templates tests failed"
popd
pushd "$KEYMAN_ROOT/common/models/types"
npm run test || fail "models/types tests failed"
popd
npm run mocha -- --recursive $_FLAGS ./unit_tests/headless/*.js ./unit_tests/headless/**/*.js
}
@ -85,6 +98,24 @@ if (( RUN_HEADLESS )); then
test-headless || fail "DOMless tests failed!"
fi
if (( RUN_BROWSERS )); then
if [[ $VERSION_ENVIRONMENT == test ]]; then
# If we are running a TeamCity test build, for now, only run BrowserStack
# tests when on a PR branch with a title including "(web)" or with the label
# test-browserstack. This is because the BrowserStack tests are currently
# unreliable, and the false positive failures are masking actual failures.
#
# We do not run BrowserStack tests on master, beta, or stable-x.y test
# builds.
RUN_BROWSERS=0
if builder_pull_get_details; then
if [[ $builder_pull_title =~ \(web\) ]] || builder_pull_has_label test-browserstack; then
RUN_BROWSERS=1
fi
fi
fi
fi
# Run browser-based tests.
if (( RUN_BROWSERS )); then
test-browsers || fail "Browser-based tests failed!"

View file

@ -141,7 +141,7 @@ module.exports = function(baseConfigParams /* the project's base configuration
// Concurrency level
// We are alloted 5 total at once from BrowserStack. That said, note that we have multiple build configs that may
// need BrowserStack-based testing simultaneously. It may be best to avoid bottlenecking on that limitation.
concurrency: 2,
concurrency: 1,
customLaunchers: FINAL_LAUNCHER_DEFS,

View file

@ -19,7 +19,7 @@
"karma-mocha-reporter": "^2.2.5",
"karma-safari-launcher": "^1.0.0",
"karma-teamcity-reporter": "^1.1.0",
"mocha": "^8.4.0",
"mocha": "^10.0.0",
"mocha-teamcity-reporter": "^4.0.0",
"sinon": "^14.0.0",
"ts-node": "^9.1.1",

View file

@ -21,7 +21,7 @@
"@keymanapp/resources-gosh": "*",
"@types/node": "^11.9.4",
"chai": "^4.3.4",
"mocha": "^8.4.0",
"mocha": "^10.0.0",
"mocha-teamcity-reporter": "^4.0.0",
"ts-node": "^9.1.1",
"typescript": "^4.5.4"

View file

@ -12,7 +12,7 @@ namespace com.keyman.text {
/**
* Indicates the device (platform) to be used for non-keystroke events,
* such as those sent to `begin postkeystroke` and `begin newcontext`
* such as those sent to `begin postkeystroke` and `begin newcontext`
* entry points.
*/
private contextDevice: utils.DeviceSpec;
@ -273,6 +273,7 @@ namespace com.keyman.text {
let totalMass = 0; // Tracks sum of non-error probabilities.
for(let pair of keyDistribution) {
if(pair.p < KEYSTROKE_EPSILON) {
totalMass += pair.p;
break;
} else if(timer && timer() >= TIMEOUT_THRESHOLD) {
// Note: it's always possible that the thread _executing_ our JS

View file

@ -19,7 +19,7 @@
"devDependencies": {
"@keymanapp/resources-gosh": "*",
"chai": "^4.3.4",
"mocha": "^8.4.0",
"mocha": "^10.0.0",
"mocha-teamcity-reporter": "^4.0.0",
"ts-node": "^9.1.1",
"typescript": "^4.5.4"

View file

@ -37,6 +37,7 @@ namespace com.keyman.keyboards {
layer: string;
displayLayer: string;
nextlayer: string;
sp?: ButtonClass;
private baseKeyEvent: text.KeyEvent;
isMnemonic: boolean = false;
@ -73,6 +74,13 @@ namespace com.keyman.keyboards {
return this.id;
}
@Enumerable
public get isPadding(): boolean {
// Does not include 9 (class: blank) as that may be an intentional 'catch' for misplaced
// keystrokes.
return this['sp'] == 10; // Button class: hidden.
}
/**
* A unique identifier based on both the key ID & the 'desktop layer' to be used for the key.
*
@ -337,14 +345,12 @@ namespace com.keyman.keyboards {
// Allow for right OSK margin (15 layout units)
let rightMargin = ActiveKey.DEFAULT_RIGHT_MARGIN/totalWidth;
totalPercent += rightMargin;
// If a single key, and padding is negative, add padding to right align the key
if(keys.length == 1 && parseInt(keys[0]['pad'],10) < 0) {
keyPercent=parseInt(keys[0]['width'],10)/totalWidth;
keys[0]['widthpc']=keyPercent;
totalPercent += keyPercent;
keys[0]['padpc']=1-totalPercent;
keys[0]['padpc']=1-(totalPercent + keyPercent + rightMargin);
// compute center's default x-coord (used in headless modes)
setProportions(keys[0] as ActiveKey, padPercent, keyPercent, totalPercent);
@ -352,8 +358,7 @@ namespace com.keyman.keyboards {
let j=keys.length-1;
padPercent=parseInt(keys[j]['pad'],10)/totalWidth;
keys[j]['padpc']=padPercent;
totalPercent += padPercent;
keys[j]['widthpc'] = keyPercent = 1-totalPercent;
keys[j]['widthpc'] = keyPercent = 1-(totalPercent + padPercent + rightMargin);
// compute center's default x-coord (used in headless modes)
setProportions(keys[j] as ActiveKey, padPercent, keyPercent, totalPercent);
@ -515,7 +520,7 @@ namespace com.keyman.keyboards {
// Should we wish to allow multiple different transforms for distance -> probability, use a function parameter in place
// of the formula in the loop below.
for(let key in keyDists) {
totalMass += keyProbs[key] = 1 / (keyDists[key] + 1e-6); // Prevent div-by-0 errors.
totalMass += keyProbs[key] = 1 / (Math.pow(keyDists[key], 2) + 1e-6); // Prevent div-by-0 errors.
}
for(let key in keyProbs) {
@ -550,6 +555,8 @@ namespace com.keyman.keyboards {
// Results in a more optimized distribution.
if(text.Codes.isKnownOSKModifierKey(key.baseKeyID)) {
return;
} else if(key.isPadding) { // to the user, blank / padding keys do not exist.
return;
}
}
// These represent the within-key distance of the touch from the key's center.

View file

@ -81,10 +81,13 @@ namespace com.keyman.text {
case 'K_SHIFT':
case 'K_LOPT':
case 'K_ROPT':
case 'K_NUMLOCK': // Often used for numeric layers.
case 'K_NUMLOCK': // Often used for numeric layers.
case 'K_CAPS':
return true;
default:
if(Codes.keyCodes[keyID] >= 50000) { // A few are used by `sil_euro_latin`.
return true; // is a 'K_' key defined for layer shifting or 'control' use.
}
// Refer to text/codes.ts - these are Keyman-custom "keycodes" used for
// layer shifting keys. To be safe, we currently let K_TABBACK and
// K_TABFWD through, though we might be able to drop them too.

View file

@ -29,7 +29,7 @@
"karma-mocha-reporter": "^2.2.5",
"karma-safari-launcher": "^1.0.0",
"karma-teamcity-reporter": "^1.1.0",
"mocha": "^8.4.0",
"mocha": "^10.0.0",
"mocha-teamcity-reporter": "^4.0.0",
"sinon": "^14.0.0",
"ts-node": "^9.1.1",

View file

@ -32,10 +32,6 @@ namespace correction {
replacements: TrackedContextSuggestion[];
activeReplacementId: number = -1;
get isNew(): boolean {
return this.transformDistributions.length == 0;
}
get currentText(): string {
if(this.replacementText === undefined || this.replacementText === null) {
return this.raw;
@ -89,7 +85,7 @@ namespace correction {
if(token.replacementText) {
copy.replacementText = token.replacementText;
}
return copy;
});
this.searchSpace = obj.searchSpace;
@ -139,8 +135,8 @@ namespace correction {
// Track the Transform that resulted in the whitespace 'token'.
// Will be needed for phrase-level correction/prediction.
whitespaceToken.transformDistributions = [transformDistribution];
whitespaceToken.transformDistributions = transformDistribution ? [transformDistribution] : [];
whitespaceToken.raw = null;
this.tokens.push(whitespaceToken);
}
@ -149,19 +145,19 @@ namespace correction {
* Used for 14.0's backspace workaround, which flattens all previous Distribution<Transform>
* entries because of limitations with direct use of backspace transforms.
* @param tokenText
* @param transformId
* @param transformId
*/
replaceTailForBackspace(tokenText: USVString, transformId: number) {
this.tokens.pop();
// It's a backspace transform; time for special handling!
//
// For now, with 14.0, we simply compress all remaining Transforms for the token into
// multiple single-char transforms. Probabalistically modeling BKSP is quite complex,
// For now, with 14.0, we simply compress all remaining Transforms for the token into
// multiple single-char transforms. Probabalistically modeling BKSP is quite complex,
// so we simplify by assuming everything remaining after a BKSP is 'true' and 'intended' text.
//
// Note that we cannot just use a single, monolithic transform at this point b/c
// of our current edit-distance optimization strategy; diagonalization is currently...
// of our current edit-distance optimization strategy; diagonalization is currently...
// not very compatible with that.
let backspacedTokenContext: Distribution<Transform>[] = textToCharTransforms(tokenText, transformId).map(function(transform) {
return [{sample: transform, p: 1.0}];
@ -175,7 +171,7 @@ namespace correction {
updateTail(transformDistribution: Distribution<Transform>, tokenText?: USVString) {
let editedToken = this.tail;
// Preserve existing text if new text isn't specified.
tokenText = tokenText || (tokenText === '' ? '' : editedToken.raw);
@ -191,7 +187,7 @@ namespace correction {
toRawTokenization() {
let sequence: USVString[] = [];
for(let token of this.tokens) {
// Hide any tokens representing wordbreaks. (Thinking ahead to phrase-level possibilities)
if(token.currentText !== null) {
@ -281,7 +277,7 @@ namespace correction {
/**
* Returns items contained within the circular array, ordered from 'oldest' to 'newest' -
* the same order in which the items will be dequeued.
* @param index
* @param index
*/
item(index: number) {
if(index >= this.count) {
@ -294,7 +290,7 @@ namespace correction {
}
export class ContextTracker extends CircularArray<TrackedContextState> {
static attemptMatchContext(tokenizedContext: USVString[],
static attemptMatchContext(tokenizedContext: USVString[],
matchState: TrackedContextState,
transformDistribution?: Distribution<Transform>,): TrackedContextState {
// Map the previous tokenized state to an edit-distance friendly version.
@ -335,7 +331,7 @@ namespace correction {
}
// Can happen for the first text input after backspace deletes a wordbreaking character,
// thus the new input continues a previous word while dropping the empty word after
// thus the new input continues a previous word while dropping the empty word after
// that prior wordbreaking character.
//
// We can't handle it reliably from this match state, but a previous entry (without the empty token)
@ -353,7 +349,7 @@ namespace correction {
// If we've made it here... success! We have a context match!
let state: TrackedContextState;
if(pushedTail) {
// On suggestion acceptance, we should update the previous final token.
// We do it first so that the acceptance is replicated in the new TrackedContextState
@ -376,7 +372,9 @@ namespace correction {
if(primaryInput && primaryInput.insert == "" && primaryInput.deleteLeft == 0 && !primaryInput.deleteRight) {
primaryInput = null;
}
const isBackspace = primaryInput && primaryInput.insert == "" && primaryInput.deleteLeft > 0 && !primaryInput.deleteRight;
const isWhitespace = primaryInput && TransformUtils.isWhitespace(primaryInput);
const isBackspace = primaryInput && TransformUtils.isBackspace(primaryInput);
const finalToken = tokenizedContext[tokenizedContext.length-1];
/* Assumption: This is an adequate check for its two sub-branches.
@ -388,7 +386,7 @@ namespace correction {
* - Assumption: one keystroke may only cause a single token to be appended to the context
* - That is, no "reasonable" keystroke would emit a Transform adding two separate word tokens
* - For languages using whitespace to word-break, said keystroke would have to include said whitespace to break the assumption.
*/
*/
// If there is/was more than one context token available...
if(editPath.length > 1) {
@ -399,17 +397,29 @@ namespace correction {
// We're adding an additional context token.
if(pushedTail) {
// ASSUMPTION: any transform that triggers this case is a pure-whitespace Transform, as we
// need a word-break before beginning a new word's context.
// Worth note: when invalid, the lm-layer already has problems in other aspects too.
state.pushWhitespaceToTail(transformDistribution);
const tokenizedTail = tokenizedContext[tokenizedContext.length - 1];
/*
* Common-case: most transforms that trigger this case are from pure-whitespace Transforms. MOST.
*
* Less-common, but noteworthy: some wordbreaks may occur without whitespace. Example:
* `"o` => ['"', 'o']. Make sure to double-check against `tokenizedContext`!
*/
let pushedToken = new TrackedContextToken();
pushedToken.raw = tokenizedTail;
let emptyToken = new TrackedContextToken();
emptyToken.raw = '';
// Continuing the earlier assumption, that 'pure-whitespace Transform' does not emit any initial characters
// for the new word (token), so the input keystrokes do not correspond to the new text token.
emptyToken.transformDistributions = [];
state.pushTail(emptyToken);
if(isWhitespace || !primaryInput) {
state.pushWhitespaceToTail(transformDistribution ?? []);
// Continuing the earlier assumption, that 'pure-whitespace Transform' does not emit any initial characters
// for the new word (token), so the input keystrokes do not correspond to the new text token.
pushedToken.transformDistributions = [];
} else {
state.pushWhitespaceToTail();
// Assumption: Since we only allow one-transform-at-a-time changes between states, we shouldn't be missing
// any metadata used to construct the new context state token.
pushedToken.transformDistributions = transformDistribution ? [transformDistribution] : [];
}
state.pushTail(pushedToken);
} else { // We're editing the final context token.
// TODO: Assumption: we didn't 'miss' any inputs somehow.
// As is, may be prone to fragility should the lm-layer's tracked context 'desync' from its host's.
@ -442,7 +452,9 @@ namespace correction {
return state;
}
static modelContextState(tokenizedContext: USVString[], lexicalModel: LexicalModel): TrackedContextState {
static modelContextState(tokenizedContext: USVString[],
transformDistribution: Distribution<Transform>,
lexicalModel: LexicalModel): TrackedContextState {
let baseTokens = tokenizedContext.map(function(entry) {
let token = new TrackedContextToken();
token.raw = entry;
@ -483,13 +495,12 @@ namespace correction {
* Compares the current, post-input context against the most recently-seen contexts from previous prediction calls, returning
* the most information-rich `TrackedContextState` possible. If a match is found, the state will be annotated with the
* input information provided to previous prediction calls and persisted correction-search calculations for re-use.
*
* @param model
* @param context
* @param mainTransform
* @param transformDistribution
*
* @param model
* @param context
* @param transformDistribution
*/
analyzeState(model: LexicalModel,
analyzeState(model: LexicalModel,
context: Context,
transformDistribution?: Distribution<Transform>): TrackedContextState {
if(!model.traverseFromRoot) {
@ -519,7 +530,7 @@ namespace correction {
//
// Assumption: as a caret needs to move to context before any actual transform distributions occur,
// this state is only reached on caret moves; thus, transformDistribution is actually just a single null transform.
let state = ContextTracker.modelContextState(tokenizedContext.left, model);
let state = ContextTracker.modelContextState(tokenizedContext.left, transformDistribution, model);
state.taggedContext = context;
this.enqueue(state);
return state;

View file

@ -204,6 +204,15 @@ namespace correction {
// TODO: might should also track diagonalWidth.
return inputString + models.SENTINEL_CODE_UNIT + matchString;
}
get isFullReplacement(): boolean {
// If the known edit-distance cost is equal to the input length, this means
// that literally every input has been full-on replaced. Thus, this is
// likely not a good 'root' to use for predictions.
//
// Logic exception: 0 cost, 0 length != a "replacement".
return this.knownCost && this.knownCost == this.priorInput.length;
}
}
class SearchSpaceTier {
@ -526,8 +535,6 @@ namespace correction {
let searchSpace = this;
let currentReturns: {[mapKey: string]: SearchNode} = {};
// JS measures time by the number of milliseconds since Jan 1, 1970.
let timeStart = Date.now();
let maxTime: number;
if(waitMillis == 0) {
maxTime = Infinity;
@ -537,6 +544,136 @@ namespace correction {
maxTime = waitMillis;
}
/**
* This inner class is designed to help the algorithm detect its active execution time.
* While there's no official JS way to do this, we can approximate it by polling the
* current system time (in ms) after each iteration of a short-duration loop. Unusual
* spikes in system time for a single iteration is likely to indicate that an OS
* context switch occurred at some point during the iteration's execution.
*/
class ExecutionTimer {
/**
* The system time when this instance was created.
*/
private start: number;
/**
* Marks the system time at the start of the currently-running loop, as noted
* by a call to the `startLoop` function.
*/
private loopStart: number;
private maxExecutionTime: number;
private maxTrueTime: number;
private executionTime: number;
/**
* Used to track intervals in which potential context swaps by the OS may
* have occurred. Context switches generally seem to pause threads for
* at least 16 ms, while we expect each loop iteration to complete
* within just 1 ms. So, any possible context switch should have the
* longest observed change in system time.
*
* See `updateOutliers` for more details.
*/
private largestIntervals: number[] = [0];
constructor(maxExecutionTime: number, maxTrueTime: number) {
// JS measures time by the number of milliseconds since Jan 1, 1970.
this.loopStart = this.start = Date.now();
this.maxExecutionTime = maxExecutionTime;
this.maxTrueTime = maxTrueTime;
}
startLoop() {
this.loopStart = Date.now();
}
markIteration() {
const now = Date.now();
const delta = now - this.loopStart;
this.executionTime += delta;
/**
* Update the list of the three longest system-time intervals observed
* for execution of a single loop iteration.
*
* Ignore any zero-ms length intervals; they'd make the logic much
* messier than necessary otherwise.
*/
if(delta) {
// If the currently-observed interval is longer than the shortest of the 3
// previously-observed longest intervals, replace it.
if(this.largestIntervals.length > 2 && delta > this.largestIntervals[0]) {
this.largestIntervals[0] = delta;
} else {
this.largestIntervals.push(delta);
}
// Puts the list in ascending order. Shortest of the list becomes the head,
// longest one the tail.
this.largestIntervals.sort();
// Then, determine if we need to update our outlier-based tweaks.
this.updateOutliers();
}
}
updateOutliers() {
/* Base assumption: since each loop of the search should evaluate within ~1ms,
* notably longer execution times are probably context switches.
*
* Base assumption: OS context switches generally last at least 16ms. (Based on
* a window.setTimeout() usually not evaluating for at least
* that long, even if set to 1ms.)
*
* To mitigate these assumptions: we'll track the execution time of every loop
* iteration. If the longest observation somehow matches or exceeds the length of
* the next two almost-longest observations twice over... we have a very strong
* 'context switch' candidate.
*
* Or, in near-formal math/stats: we expect a very low variance in execution
* time among the iterations of the search's loops. With a very low variance,
* ANY significant proportional spikes in execution time are outliers - outliers
* likely caused by an OS context switch.
*
* Rather than do intensive math, we use a somewhat lazy approach below that
* achieves the same net results given our assumptions, even when relaxed somewhat.
*
* The logic below relaxes the base assumptions a bit to be safe:
* - [2ms, 2ms, 8ms] will cause 8ms to be seen as an outlier.
* - [2ms, 3ms, 10ms] will cause 10ms to be seen as an outlier.
*
* Ideally:
* - [1ms, 1ms, 4ms] will view 4ms as an outlier.
*
* So we can safely handle slightly longer average intervals and slightly shorter
* OS context-switch time intervals.
*/
if(this.largestIntervals.length > 2) {
// Precondition: the `largestIntervals` array is sorted in ascending order.
// Shortest entry is at the head, longest at the tail.
if(this.largestIntervals[2] >= 2 * (this.largestIntervals[0] + this.largestIntervals[1])) {
this.executionTime -= this.largestIntervals[2];
this.largestIntervals.pop();
}
}
}
shouldTimeout(): boolean {
const now = Date.now();
if(now - this.start > this.maxTrueTime) {
return true;
}
return this.executionTime > this.maxExecutionTime;
}
resetOutlierCheck() {
this.largestIntervals = [];
}
}
class BatchingAssistant {
currentCost = Number.MIN_SAFE_INTEGER;
entries: SearchResult[] = [];
@ -582,17 +719,30 @@ namespace correction {
let batcher = new BatchingAssistant();
const timer = new ExecutionTimer(maxTime*1.5, maxTime);
// Stage 1 - if we already have extracted results, build a queue just for them and iterate over it first.
let returnedValues = Object.values(this.returnedValues);
if(returnedValues.length > 0) {
let preprocessedQueue = new models.PriorityQueue<SearchNode>(QUEUE_NODE_COMPARATOR, returnedValues);
// Build batches of same-cost entries.
timer.startLoop();
while(preprocessedQueue.count > 0) {
let entry = preprocessedQueue.dequeue();
// Is the entry a reasonable result?
if(entry.isFullReplacement) {
// If the entry's 'match' fully replaces the input string, we consider it
// unreasonable and ignore it.
continue;
}
let batch = batcher.checkAndAdd(entry);
timer.markIteration();
if(batch) {
// Do not track yielded time.
yield batch;
}
}
@ -601,11 +751,14 @@ namespace correction {
// finalize the last preprocessed group without issue.
let batch = batcher.tryFinalize();
if(batch) {
// Do not track yielded time.
yield batch;
}
}
// Stage 2: the fun part; actually searching!
timer.resetOutlierCheck();
timer.startLoop();
let timedOut = false;
do {
let newResult: PathResult;
@ -613,10 +766,9 @@ namespace correction {
// Search for a 'complete' path, skipping all partial paths as long as time remains.
do {
newResult = this.handleNextNode();
timer.markIteration();
// (Naive) timeout check!
let now = Date.now();
if(now - timeStart > maxTime) {
if(timer.shouldTimeout()) {
timedOut = true;
}
} while(!timedOut && newResult.type == 'intermediate')
@ -626,6 +778,13 @@ namespace correction {
if(newResult.type == 'none') {
break;
} else if(newResult.type == 'complete') {
// Is the entry a reasonable result?
if(newResult.finalNode.isFullReplacement) {
// If the entry's 'match' fully replaces the input string, we consider it
// unreasonable and ignore it. Also, if we've reached this point...
// we can(?) assume that everything thereafter is as well.
break;
}
batch = batcher.checkAndAdd(newResult.finalNode);
}

View file

@ -32,6 +32,7 @@
/// <reference types="@keymanapp/lm-message-types" />
/// <reference path="./models/dummy-model.ts" />
/// <reference path="./model-compositor.ts" />
/// <reference path="./transformUtils.ts" />
/**
* Encapsulates all the state required for the LMLayer's worker thread.
@ -407,6 +408,7 @@ if (typeof module !== 'undefined' && typeof module.exports !== 'undefined') {
module.exports['wordBreakers'] = wordBreakers;
/// XXX: export the ModelCompositor for testing.
module.exports['ModelCompositor'] = ModelCompositor;
module.exports['TransformUtils'] = TransformUtils;
} else if (typeof self !== 'undefined' && 'postMessage' in self && 'importScripts' in self) {
// Automatically install if we're in a Web Worker.
LMLayerWorker.install(self as any); // really, 'as typeof globalThis', but we're currently getting TS errors from use of that.

View file

@ -6,6 +6,23 @@ class ModelCompositor {
private static readonly MAX_SUGGESTIONS = 12;
readonly punctuation: LexicalModelPunctuation;
/**
* Controls the strength of anti-corrective measures for single-character scenarios.
* The base key probability will be raised to this power for this specific case.
*
* Current selection's motivation: (0.5 / 0.4) ^ 16 ~= 35.5.
* - if the most likely has p=0.5 and second-most has p=0.4 - a highly-inaccurate key
* stroke - the net effect will apply a factor of 35.5 to the lexical probability of
* the best key's prediction roots, favoring it in this manner.
* - less extreme edge cases will have a significantly stronger factor, acting as a
* "soft threshold".
* - truly ambiguous, "coin flip" cases will have a lower factor and thus favor the
* more likely words from the pair.
* - Our OSK key-element borders aren't visible to the user, so the 'spot' where
* behavior changes might feel arbitrary to users if we used a hard threshold instead.
*/
private static readonly SINGLE_CHAR_KEY_PROB_EXPONENT = 16;
private SUGGESTION_ID_SEED = 0;
constructor(lexicalModel: LexicalModel) {
@ -16,30 +33,6 @@ class ModelCompositor {
this.punctuation = ModelCompositor.determinePunctuationFromModel(lexicalModel);
}
protected isWhitespace(transform: Transform): boolean {
// Matches prefixed text + any instance of a character with Unicode general property Z* or the following: CR, LF, and Tab.
let whitespaceRemover = /.*[\u0009\u000A\u000D\u0020\u00a0\u1680\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u200b\u2028\u2029\u202f\u205f\u3000]/i;
// Filter out null-inserts; their high probability can cause issues.
if(transform.insert == '') { // Can actually register as 'whitespace'.
return false;
}
let insert = transform.insert;
insert = insert.replace(whitespaceRemover, '');
return insert == '';
}
protected isBackspace(transform: Transform): boolean {
return transform.insert == "" && transform.deleteLeft > 0;
}
protected isEmpty(transform: Transform): boolean {
return transform.insert == '' && transform.deleteLeft == 0;
}
private predictFromCorrections(corrections: ProbabilityMass<Transform>[], context: Context): Distribution<Suggestion> {
let returnedPredictions: Distribution<Suggestion> = [];
@ -98,8 +91,8 @@ class ModelCompositor {
})[0].sample;
// Only allow new-word suggestions if space was the most likely keypress.
let allowSpace = this.isWhitespace(inputTransform);
let allowBksp = this.isBackspace(inputTransform);
let allowSpace = TransformUtils.isWhitespace(inputTransform);
let allowBksp = TransformUtils.isBackspace(inputTransform);
let postContext = models.applyTransform(inputTransform, context);
let keepOptionText = this.wordbreak(postContext);
@ -109,7 +102,7 @@ class ModelCompositor {
// Used to restore whitespaces if operations would remove them.
let prefixTransform: Transform;
let contextState: correction.TrackedContextState = null;
let postContextState: correction.TrackedContextState = null;
// Section 1: determining 'prediction roots'.
if(!this.contextTracker) {
@ -124,18 +117,18 @@ class ModelCompositor {
predictionRoots = [{sample: inputTransform, p: 1.0}];
prefixTransform = inputTransform;
} else {
predictionRoots = transformDistribution.map(function(alt) {
predictionRoots = transformDistribution.map((alt) => {
let transform = alt.sample;
// Filter out special keys unless they're expected.
if(this.isWhitespace(transform) && !allowSpace) {
if(TransformUtils.isWhitespace(transform) && !allowSpace) {
return null;
} else if(this.isBackspace(transform) && !allowBksp) {
} else if(TransformUtils.isBackspace(transform) && !allowBksp) {
return null;
}
return alt;
}, this);
});
}
// Remove `null` entries.
@ -144,12 +137,15 @@ class ModelCompositor {
// Running in bulk over all suggestions, duplicate entries may be possible.
rawPredictions = this.predictFromCorrections(predictionRoots, context);
} else {
contextState = this.contextTracker.analyzeState(this.lexicalModel,
postContext,
!this.isEmpty(inputTransform) ?
transformDistribution:
null
);
// Token replacement benefits greatly from knowledge of the prior context state.
let contextState = this.contextTracker.analyzeState(this.lexicalModel, context, null);
// Corrections and predictions are based upon the post-context state, though.
postContextState = this.contextTracker.analyzeState(this.lexicalModel,
postContext,
!TransformUtils.isEmpty(inputTransform) ?
transformDistribution:
null
);
// TODO: Should we filter backspaces & whitespaces out of the transform distribution?
// Ideally, the answer (in the future) will be no, but leaving it in right now may pose an issue.
@ -158,19 +154,81 @@ class ModelCompositor {
// let's just note that right now, there will only ever be one.
//
// The 'eventual' logic will be significantly more complex, though still manageable.
let searchSpace = contextState.searchSpace[0];
let searchSpace = postContextState.searchSpace[0];
let newEmptyToken = false;
// Detect if we're starting a new context state.
let contextTokens = contextState.tokens;
if(contextTokens.length == 0 || contextTokens[contextTokens.length - 1].isNew) {
if(this.isEmpty(inputTransform) || this.isWhitespace(inputTransform)) {
newEmptyToken = true;
// No matter the prediction, once we know the root of the prediction, we'll always 'replace' the
// same amount of text. We can handle this before the big 'prediction root' loop.
let deleteLeft = 0;
// The amount of text to 'replace' depends upon whatever sort of context change occurs
// from the received input.
const postContextTokens = postContextState.tokens;
let postContextLength = postContextTokens.length;
let contextLengthDelta = postContextTokens.length - contextState.tokens.length;
// If the context now has more tokens, the token we'll be 'predicting' didn't originally exist.
if(postContextLength == 0 || contextLengthDelta > 0) {
// As the word/token being corrected/predicted didn't originally exist, there's no
// part of it to 'replace'.
deleteLeft = 0;
// If the new token is due to whitespace or due to a different input type that would
// likely imply a tokenization boundary...
if(TransformUtils.isWhitespace(inputTransform)) {
/* TODO: consider/implement: the second half of the comment above.
* For example: on input of a `'`, predict new words instead of replacing the `'`.
* (since after a letter, the `'` will be ignored, anyway)
*
* Idea: if the model's most likely prediction (with no root) would make a new
* token if appended to the current token, that's probably a good case.
* Keeps the check simple & quick.
*
* Might need a mixed mode, though: ';' is close enough that `l` is a reasonable
* fat-finger guess. So yeah, we're not addressing this idea right now.
* - so... consider multiple context behavior angles when building prediction roots?
*
* May need something similar to help handle contractions during their construction,
* but that'd be within `ContextTracker`.
* can' => [`can`, `'`]
* can't => [`can't`] (WB6, 7 of https://unicode.org/reports/tr29/#Word_Boundary_Rules)
*
* (Would also helps WB7b+c for Hebrew text)
*/
// Infer 'new word' mode, even if we received new text when reaching
// this position. That new text didn't exist before, so still - nothing
// to 'replace'.
prefixTransform = inputTransform;
context = postContext; // Ensure the whitespace token is preapplied!
context = postContext; // As far as predictions are concerned, the post-context state
// should not be replaced. Predictions are to be rooted on
// text "up for correction" - so we want a null root for this
// branch.
contextState = postContextState;
}
// If the tokenized context length is shorter... sounds like a backspace (or similar).
} else if (contextLengthDelta < 0) {
/* Ooh, we've dropped context here. Almost certainly from a backspace.
* Even if we drop multiple tokens... well, we know exactly how many chars
* were actually deleted - `inputTransform.deleteLeft`.
* Since we replace a word being corrected/predicted, we take length of the remaining
* context's tail token in addition to however far was deleted to reach that state.
*/
deleteLeft = this.wordbreak(postContext).kmwLength() + inputTransform.deleteLeft;
} else {
// Suggestions are applied to the pre-input context, so get the token's original length.
// We're on the same token, so just delete its text for the replacement op.
deleteLeft = this.wordbreak(context).kmwLength();
}
// Is the token under construction newly-constructed / is there no pre-existing root?
// If so, we want to strongly avoid overcorrection, even for 'nearby' keys.
// (Strong lexical frequency differences can easily cause overcorrection when only
// one key's available.)
//
// NOTE: we only want this applied word-initially, when any corrections 'correct'
// 100% of the word. Things are generally fine once it's not "all or nothing."
let tailToken = postContextTokens[postContextTokens.length - 1];
const isTokenStart = tailToken.transformDistributions.length <= 1;
// TODO: whitespace, backspace filtering. Do it here.
// Whitespace is probably fine, actually. Less sure about backspace.
@ -192,19 +250,6 @@ class ModelCompositor {
finalInput = inputTransform; // A fallback measure. Greatly matters for empty contexts.
}
let deleteLeft = 0;
// remove actual token string. If new token, there should be nothing to delete.
if(!newEmptyToken) {
// If this is triggered from a backspace, make sure to use its results
// and also include its left-deletions! It's the one post-input context case.
if(allowBksp) {
deleteLeft = this.wordbreak(postContext).kmwLength() + inputTransform.deleteLeft;
} else {
// Normal case - use the pre-input context.
deleteLeft = this.wordbreak(context).kmwLength();
}
}
// Replace the existing context with the correction.
let correctionTransform: Transform = {
insert: correction, // insert correction string
@ -212,9 +257,39 @@ class ModelCompositor {
id: inputTransform.id // The correction should always be based on the most recent external transform/transcription ID.
}
let rootCost = match.totalCost;
/* If we're dealing with the FIRST keystroke of a new sequence, we'll **dramatically** boost
* the exponent to ensure only VERY nearby corrections have a chance of winning, and only if
* there are significantly more likely words. We only need this to allow very minor fat-finger
* adjustments for 100% keystroke-sequence corrections in order to prevent finickiness on
* key borders.
*
* Technically, the probabilities this produces won't be normalized as-is... but there's no
* true NEED to do so for it, even if it'd be 'nice to have'. Consistently tracking when
* to apply it could become tricky, so it's simpler to leave out.
*
* Worst-case, it's possible to temporarily add normalization if a code deep-dive
* is needed in the future.
*/
if(isTokenStart) {
/* Suppose a key distribution: most likely with p=0.5, second-most with 0.4 - a pretty
* ambiguous case that would only arise very near the center of the boundary between two keys.
* Raising (0.5/0.4)^16 ~= 35.53. (At time of writing, SINGLE_CHAR_KEY_PROB_EXPONENT = 16.)
* That seems 'within reason' for correction very near boundaries.
*
* So, with the second-most-likely key being that close in probability, its best suggestion
* must be ~ 35.5x more likely than that of the truly-most-likely key to "win". So, it's not
* a HARD cutoff, but more of a 'soft' one. Keeping the principles in mind documented above,
* it's possible to tweak this to a more harsh or lenient setting if desired, rather than
* being totally "all or nothing" on which key is taken for highly-ambiguous keypresses.
*/
rootCost *= ModelCompositor.SINGLE_CHAR_KEY_PROB_EXPONENT; // note the `Math.exp` below.
}
return {
sample: correctionTransform,
p: Math.exp(-match.totalCost)
p: Math.exp(-rootCost)
};
}, this);
@ -411,8 +486,8 @@ class ModelCompositor {
// Store the suggestions on the final token of the current context state (if it exists).
// Or, once phrase-level suggestions are possible, on whichever token serves as each prediction's root.
if(contextState) {
contextState.tail.replacements = suggestions.map(function(suggestion) {
if(postContextState) {
postContextState.tail.replacements = suggestions.map(function(suggestion) {
return {
suggestion: suggestion,
tokenWidth: 1
@ -659,7 +734,7 @@ class ModelCompositor {
// than before.
if(this.contextTracker) {
let tokenizedContext = models.tokenize(this.lexicalModel.wordbreaker || wordBreakers.default, context);
let contextState = correction.ContextTracker.modelContextState(tokenizedContext.left, this.lexicalModel);
let contextState = correction.ContextTracker.modelContextState(tokenizedContext.left, null, this.lexicalModel);
this.contextTracker.enqueue(contextState);
}
}

View file

@ -0,0 +1,15 @@
class TransformUtils {
static isWhitespace(transform: Transform): boolean {
// Matches a string that is entirely one or more characters with Unicode general property Z* or the following: CR, LF, and Tab.
const whitespaceRemover = /^[\u0009\u000A\u000D\u0020\u00a0\u1680\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u200b\u2028\u2029\u202f\u205f\u3000]+$/i;
return transform.insert.match(whitespaceRemover) != null;
}
static isBackspace(transform: Transform): boolean {
return transform.insert == "" && transform.deleteLeft > 0 && !transform.deleteRight;
}
static isEmpty(transform: Transform): boolean {
return transform.insert == '' && transform.deleteLeft == 0 && !transform.deleteRight;
}
}

View file

@ -1,8 +1,8 @@
{
"name": "@keymanapp/web-utils",
"description": "Common utility functions used throughout other Keyman packages",
"main": "./dist/index.js",
"types": "./dist/index.d.ts",
"main": "./build/index.js",
"types": "./build/index.d.ts",
"scripts": {
"build": "gosh ./build.sh",
"tsc": "tsc"

0
common/windows/cef-checkout.sh Normal file → Executable file
View file

View file

@ -108,8 +108,8 @@ const
// Alpha versions will work against the staging server so that they
// can access new APIs etc that will only be available there. The staging
// servers have resource constraints but should be okay for limited use.
S_KeymanCom_Staging = 'https://keyman-staging.com';
S_APIServer_Staging = 'api.keyman-staging.com';
S_KeymanCom_Staging = 'https://keyman.com'; // #7227 disabling: 'https://keyman-staging.com';
S_APIServer_Staging = 'api.keyman.com'; // #7227 disabling: 'api.keyman-staging.com';
const
URLPath_PackageDownload_Format = '/go/package/download/%0:s?platform=windows&tier=%1:s&bcp47=%2:s&update=%3:d';

View file

@ -82,27 +82,38 @@ var
fs: TFileStream;
ms: TMemoryStream;
begin
fs := TFileStream.Create(ParamStr(0), fmOpenRead or fmShareDenyWrite);
ms := TMemoryStream.Create;
try
if not FindFirstHeader(fs) then
Exit(False);
fs.Seek(StartOfFile, TSeekOrigin.soBeginning);
ms.CopyFrom(fs, fs.Size - StartOfFile);
ms.Position := 0;
with TZipFile.Create do
fs := TFileStream.Create(ParamStr(0), fmOpenRead or fmShareDenyWrite);
ms := TMemoryStream.Create;
try
Open(ms, zmRead);
ExtractAll(ExtPath);
if not FindFirstHeader(fs) then
Exit(False);
fs.Seek(StartOfFile, TSeekOrigin.soBeginning);
ms.CopyFrom(fs, fs.Size - StartOfFile);
ms.Position := 0;
with TZipFile.Create do
try
Open(ms, zmRead);
ExtractAll(ExtPath);
finally
Free;
end;
finally
Free;
fs.Free;
ms.Free;
end;
except
on E:Exception do
begin
raise Exception.Create(
'Failed to extract setup archive. '+
'You may have run out of disk space or there may be a '+
'problem with the source files.'#13#10#13#10+
'The error received was: '+E.Message);
end;
finally
fs.Free;
ms.Free;
end;
Result := True;
end;

0
common/windows/mkver.sh Normal file → Executable file
View file

View file

@ -17,7 +17,7 @@ if hotdoc.found()
configuration: cfg)
deps = files(
'../include/keyman/keyboardprocessor.h.in',
'../src/json.hpp',
'../src/jsonpp.hpp',
'../src/utfcodec.hpp'
)

View file

@ -1118,6 +1118,49 @@ KMN_API
km_kbp_status
km_kbp_process_queued_actions(km_kbp_state *state);
/*
```
### `km_kbp_event`
##### Description:
Tell the keyboard processor that an external event has occurred, such as a keyboard
being activated through the language switching UI.
##### Return status:
- `KM_KBP_STATUS_OK`: On success.
- `KM_KBP_STATUS_NO_MEM`:
In the event memory is unavailable to allocate internal buffers.
- `KM_KBP_STATUS_INVALID_ARGUMENT`:
In the event the `state` pointer is null or an invalid event or data is passed.
The keyboard processor may generate actions which should be processed by the
consumer of the API.
The action list will be cleared at the start of this call; options and context in
the state may also be modified.
##### Parameters:
- __state__: A pointer to the opaque state object.
- __event__: The event to be processed, from km_kbp_event_code enumeration
- __data__: Additional event-specific data. Currently unused, must be nullptr.
```c
*/
KMN_API
km_kbp_status
km_kbp_event(
km_kbp_state *state,
uint32_t event,
void* data
);
enum km_kbp_event_code {
/**
* A keyboard has been activated by the user. The processor may use this
* event, for example, to switch caps lock state or provide other UX.
*/
KM_KBP_EVENT_KEYBOARD_ACTIVATED = 1,
//future: KM_KBP_EVENT_KEYBOARD_DEACTIVATED = 2,
};
#if defined(__cplusplus)
} // extern "C"
#endif

View file

@ -1,573 +0,0 @@
# Copyright: © 2018 SIL International.
# Description: Lowlevel C style python API. Intended to be wrapped by a more
# Pythonic higher level API.
# Create Date: 18 Oct 2018
# Authors: Tim Eves (TSE)
#
import ctypes
import ctypes.util
import operator
import os
from enum import auto, IntEnum, IntFlag
from ctypes import (c_uint8,
c_uint16, c_uint32,
c_size_t,
c_void_p, c_char_p,
Structure, Union, POINTER, CFUNCTYPE)
from typing import Any, Tuple
CP = c_uint16
USV = c_uint32
VirtualKey = c_uint16
libpath = os.environ.get('PYKMNKBD_LIBRARY_PATH',
ctypes.util.find_library("kmnkbp0"))
libkbp = ctypes.cdll.LoadLibrary(libpath)
# Error handling
# ==============
class StatusCode(IntEnum):
OK = 0
NO_MEM = auto()
IO_ERROR = auto()
INVALID_ARGUMENT = auto()
KEY_ERROR = auto()
OS_ERROR = 0x80000000
Status = c_uint32
def __map_oserror(code: Status) -> str:
code = StatusCode.OS_ERROR ^ code
msg = '{0!s}: OS error code (' + code + '): ' + os.strerror(code)
return msg, lambda m: OSError(code, m)
__exceptions_map = [
(None, '{0!s}: Success'),
(MemoryError, '{0!s}: memory allocation failed.'),
(RuntimeError, '{0!s}: IO Error: {1!s}'),
(ValueError, '{0!s}: Invalid argument passed.'),
(LookupError, '{0!s}: Item does not exist in: {1!s}'),
(OSError, __map_oserror)]
def status_code(code: Status, func, args):
if code == StatusCode.OK: return args
exc, msg = __exceptions_map[code]
if callable(msg): msg = msg(code)
raise exc(msg.format(libkbp._name, *args))
def null_check(code, func, args):
if code is not None: return args
raise KeyError(args[1] + ': Not found in collection')
class Dir(IntFlag):
IN = auto()
OUT = auto()
OPT = auto()
def __method(iface: str, method: str, result,
*args: Tuple[Any, Dir, str], **kwds):
proto = CFUNCTYPE(result, *map(operator.itemgetter(0), args))
params = tuple(a[1:] for a in args)
c_name = iface+'_'+method if iface else method
f = proto(('km_kbp_' + c_name, libkbp), params)
if 'errcheck' in kwds: f.errcheck = kwds.get('errcheck')
globals()[c_name] = f
# Context processing
# ==================
Context_p = c_void_p
class ContextType(IntEnum):
END = 0
CHAR = auto()
MARKER = auto()
class ContextItem(Structure):
class __ContextValue(Union):
_fields_ = (('character', USV),
('marker', c_uint32))
_anonymous_ = ('value',)
_fields_ = (('type', c_uint8),
('value', __ContextValue))
__method('context_items', 'from_utf16', Status,
(c_void_p, Dir.IN, 'text'),
(POINTER(POINTER(ContextItem)), Dir.OUT, 'context_items'),
errcheck=status_code)
__method('context_items', 'to_utf16', c_size_t,
(POINTER(ContextItem), Dir.IN, 'context_items'),
(c_void_p, Dir.IN | Dir.OPT, 'buffer'),
(c_size_t, Dir.IN | Dir.OPT, 'buffer_size'))
__method('context_items', 'dispose', None,
(POINTER(ContextItem), Dir.IN, 'context_items'))
__method('context', 'set', Status,
(Context_p, Dir.IN, 'context'),
(POINTER(ContextItem), Dir.IN, 'context_items'),
errcheck=status_code)
__method('context', 'get', POINTER(ContextItem),
(Context_p, Dir.IN, 'context'))
__method('context', 'clear', None, (Context_p, Dir.IN, 'context'))
__method('context', 'length', c_size_t, (Context_p, Dir.IN, 'context'))
__method('context', 'append', Status,
(Context_p, Dir.IN, 'context'),
(POINTER(ContextItem), Dir.IN, 'context_items'),
errcheck=status_code)
__method('context', 'shrink', Status,
(Context_p, Dir.IN, 'context'),
(c_size_t, Dir.IN, 'num'),
(POINTER(ContextItem), Dir.IN, 'prefix'),
errcheck=status_code)
class ActionItem(Structure):
class __ActionItem(Union):
_fields_ = (('marker', c_size_t),
('option', c_char_p),
('character', USV),
('vkey', VirtualKey))
_anonymous_ = ('data',)
_fields_ = (('type', c_uint8),
('reserved', c_uint8*3),
('data', __ActionItem))
class ActionType(IntEnum):
END = 0 # Marks end of action items list.
CHAR = 1
MARKER = 2 # correlates to kmn's "deadkey" markers
ALERT = 3
BACK = 4
PERSIST_OPT = 5
RESET_OPT = 6
VKEYDOWN = 7
VKEYUP = 8
VSHIFTDOWN = 9
VSHIFTUP = 10
MAX_TYPE_ID = auto()
# Option processing
# =================
OptionSet_p = c_void_p
class OptionScope(IntEnum):
UNKNOWN = auto()
KEYBOARD = auto()
ENVIRONMENT = auto()
class Option(Structure):
_fields_ = (('key', c_char_p),
('value', c_char_p),
('scope', c_uint8))
Option.END = Option(None, None)
__method('options_set', 'size', c_size_t, (OptionSet_p, Dir.IN, 'opts'))
__method('options_set', 'lookup', POINTER(Option),
(OptionSet_p, Dir.IN, 'opts'),
(c_char_p, Dir.IN, 'key'),
errcheck=null_check)
__method('options_set', 'update', Status,
(OptionSet_p, Dir.IN, 'opts'),
(POINTER(Option), Dir.IN, 'new_opts'),
errcheck=status_code)
__method('options_set', 'to_json', Status,
(OptionSet_p, Dir.IN, 'opts'),
(c_char_p, Dir.IN | Dir.OPT, 'buffer'),
(c_size_t, Dir.IN | Dir.OUT, 'space'),
errcheck=status_code)
# Keyboards
# =========
Keyboard_p = c_void_p
class KeyboardAttrs(Structure):
_fields_ = (('version_string', c_char_p),
('id', c_char_p),
('folder_path', c_char_p),
('default_options', OptionSet_p))
__method('keyboard', 'load', Status,
(c_char_p, Dir.IN, 'path'),
(POINTER(Keyboard_p), Dir.OUT, 'kb'),
errcheck=status_code)
__method('keyboard', 'dispose', None, (Keyboard_p, Dir.IN, 'kb'))
__method('keyboard', 'get_attrs', POINTER(KeyboardAttrs),
(Keyboard_p, Dir.IN, 'keyboard'))
# State processing
# ================
State_p = c_void_p
__method('state', 'create', Status,
(Keyboard_p, Dir.IN, 'keyboard'),
(POINTER(Option), Dir.IN, 'env',),
(POINTER(State_p), Dir.OUT, 'out'),
errcheck=status_code)
__method('state', 'clone', Status,
(State_p, Dir.IN, 'state'),
(POINTER(State_p), Dir.OUT, 'out'),
errcheck=status_code)
__method('state', 'dispose', None, (State_p, Dir.IN, 'state'))
__method('state', 'context', Context_p, (State_p, Dir.IN, 'state'))
__method('state', 'options', OptionSet_p, (State_p, Dir.IN, 'state'))
__method('state', 'action_items', POINTER(ActionItem),
(State_p, Dir.IN, 'state'),
(POINTER(c_size_t), Dir.OUT, 'num_items'))
__method('state', 'to_json', Status,
(State_p, Dir.IN, 'state'),
(c_char_p, Dir.IN | Dir.OPT, 'buffer'),
(c_size_t, Dir.IN | Dir.OUT, 'space'),
errcheck=status_code)
# Processor
# =========
class Attributes(Structure):
_fields_ = (('max_context', c_size_t),
('current', c_uint16),
('revision', c_uint16),
('age', c_uint16),
('technology', c_uint16),
('vendor', c_char_p))
class Tech(IntFlag):
UNSPECIFIED = 0
KMN = 1
LDML = 2
__method(None, 'get_engine_attrs', POINTER(Attributes))
__method(None, 'process_event', Status,
(State_p, Dir.IN, 'state'),
(VirtualKey, Dir.IN, 'vkey'),
(c_uint16, Dir.IN, 'modifier_state'))
class Modifier(IntFlag):
LCTRL = 1 << 0
RCTRL = 1 << 1
LALT = 1 << 2
RALT = 1 << 3
SHIFT = 1 << 4
CTRL = 1 << 5
ALT = 1 << 6
CAPS = 1 << 7
NOCAPS = 1 << 8
NUMLOCK = 1 << 9
NONUMLOCK = 1 << 10
SCROLLOCK = 1 << 11
NOSCROLLOCK = 1 << 12
VIRTUALKEY = 1 << 13
class ModifierMask(IntFlag):
ALL = 0x7f
ALT_GR_SIM = Modifier.LCTRL | Modifier.LALT
CHIRAL = 0x1f
IS_CHIRAL = 0x0f
NON_CHIRAL = 0x7f
CAPS = 0x0300
NUMLOCK = 0x0C00
SCROLLLOCK = 0x3000
class VKey(IntEnum):
_00 = auto()
LBUTTON = auto()
RBUTTON = auto()
CANCEL = auto()
MBUTTON = auto()
_05 = auto()
_06 = auto()
_07 = auto()
BKSP = auto()
KTAB = auto()
_0A = auto()
_0B = auto()
KP5 = auto()
ENTER = auto()
_0E = auto()
_0F = auto()
SHIFT = auto()
CONTROL = auto()
ALT = auto()
PAUSE = auto()
CAPS = auto()
_15 = auto()
_16 = auto()
_17 = auto()
_18 = auto()
_19 = auto()
_1A = auto()
ESC = auto()
_1C = auto()
_1D = auto()
_1E = auto()
_1F = auto()
SPACE = auto()
PGUP = auto()
PGDN = auto()
END = auto()
HOME = auto()
LEFT = auto()
UP = auto()
RIGHT = auto()
DOWN = auto()
SEL = auto()
PRINT = auto()
EXEC = auto()
PRTSCN = auto()
INS = auto()
DEL = auto()
HELP = auto()
K0 = auto()
K1 = auto()
K2 = auto()
K3 = auto()
K4 = auto()
K5 = auto()
K6 = auto()
K7 = auto()
K8 = auto()
K9 = auto()
_3A = auto()
_3B = auto()
_3C = auto()
_3D = auto()
_3E = auto()
_3F = auto()
_40 = auto()
KA = auto()
KB = auto()
KC = auto()
KD = auto()
KE = auto()
KF = auto()
KG = auto()
KH = auto()
KI = auto()
KJ = auto()
KK = auto()
KL = auto()
KM = auto()
KN = auto()
KO = auto()
KP = auto()
KQ = auto()
KR = auto()
KS = auto()
KT = auto()
KU = auto()
KV = auto()
KW = auto()
KX = auto()
KY = auto()
KZ = auto()
_5B = auto()
_5C = auto()
_5D = auto()
_5E = auto()
_5F = auto()
NP0 = auto()
NP1 = auto()
NP2 = auto()
NP3 = auto()
NP4 = auto()
NP5 = auto()
NP6 = auto()
NP7 = auto()
NP8 = auto()
NP9 = auto()
NPSTAR = auto()
NPPLUS = auto()
SEPARATOR = auto()
NPMINUS = auto()
NPDOT = auto()
NPSLASH = auto()
F1 = auto()
F2 = auto()
F3 = auto()
F4 = auto()
F5 = auto()
F6 = auto()
F7 = auto()
F8 = auto()
F9 = auto()
F10 = auto()
F11 = auto()
F12 = auto()
F13 = auto()
F14 = auto()
F15 = auto()
F16 = auto()
F17 = auto()
F18 = auto()
F19 = auto()
F20 = auto()
F21 = auto()
F22 = auto()
F23 = auto()
F24 = auto()
_88 = auto()
_89 = auto()
_8A = auto()
_8B = auto()
_8C = auto()
_8D = auto()
_8E = auto()
_8F = auto()
NUMLOCK = auto()
SCROLL = auto()
_92 = auto()
_93 = auto()
_94 = auto()
_95 = auto()
_96 = auto()
_97 = auto()
_98 = auto()
_99 = auto()
_9A = auto()
_9B = auto()
_9C = auto()
_9D = auto()
_9E = auto()
_9F = auto()
_A0 = auto()
_A1 = auto()
_A2 = auto()
_A3 = auto()
_A4 = auto()
_A5 = auto()
_A6 = auto()
_A7 = auto()
_A8 = auto()
_A9 = auto()
_AA = auto()
_AB = auto()
_AC = auto()
_AD = auto()
_AE = auto()
_AF = auto()
_B0 = auto()
_B1 = auto()
_B2 = auto()
_B3 = auto()
_B4 = auto()
_B5 = auto()
_B6 = auto()
_B7 = auto()
_B8 = auto()
_B9 = auto()
COLON = auto()
EQUAL = auto()
COMMA = auto()
HYPHEN = auto()
PERIOD = auto()
SLASH = auto()
BKQUOTE = auto()
_C1 = auto()
_C2 = auto()
_C3 = auto()
_C4 = auto()
_C5 = auto()
_C6 = auto()
_C7 = auto()
_C8 = auto()
_C9 = auto()
_CA = auto()
_CB = auto()
_CC = auto()
_CD = auto()
_CE = auto()
_CF = auto()
_D0 = auto()
_D1 = auto()
_D2 = auto()
_D3 = auto()
_D4 = auto()
_D5 = auto()
_D6 = auto()
_D7 = auto()
_D8 = auto()
_D9 = auto()
_DA = auto()
LBRKT = auto()
BKSLASH = auto()
RBRKT = auto()
QUOTE = auto()
oDF = auto()
oE0 = auto()
oE1 = auto()
oE2 = auto()
oE3 = auto()
oE4 = auto()
_E5 = auto()
oE6 = auto()
_E7 = auto()
_E8 = auto()
oE9 = auto()
oEA = auto()
oEB = auto()
oEC = auto()
oED = auto()
oEE = auto()
oEF = auto()
oF0 = auto()
oF1 = auto()
oF2 = auto()
oF3 = auto()
oF4 = auto()
oF5 = auto()
_F6 = auto()
_F7 = auto()
_F8 = auto()
_F9 = auto()
_FA = auto()
_FB = auto()
_FC = auto()
_FD = auto()
_FE = auto()
_FF = auto()

View file

@ -1,118 +0,0 @@
import pathlib
from typing import NamedTuple, NewType, Tuple, List, Union
from enum import Enum
from collections.abc import Sequence, Mapping
USV = int
Marker = int
VirtualKey = int
class Context(Sequence):
Item = Union[USV, Marker]
def __init__(initial: str):
pass
def __del__(self):
pass
def __str__(self):
pass
def set(self, ctxt: List[Item]):
self.clear()
self.apped(ctxt)
def clear(self):
pass
def __getitem__(self, key: int) -> Item:
pass
def __len__(self):
pass
def append(self, ctxt: List[Item]):
pass
def delete(self, remove_n: int, prefix: List[Item]):
pass
Option = Tuple[str, str]
class OptionSet(Mapping):
def __getitem__(self, key: str) -> Option:
pass
def __iter__(self):
pass
def __len__(self):
pass
def __str__(self):
pass
class Keyboard(NamedTuple('__kb_attrs', version=str, id=str, folder_path=pathlib.Path, default_options=OptionSet)):
def __new__(cls, _):
return super(Keyboard, cls).__new__(cls, *[None]*4)
def __init__(self, kb_path: pathlib.Path):
pass
def __del__(self):
pass
class Action:
VKeyDown = NewType('Action.VirtualKey', VirtualKey)
# VKeyUp = VirtualKey
# VShiftDown = VirtualKey
# VShiftUp = VirtualKey
# Char = int
# Marker = int
# Bell = NewType('Bell', None)
# Back = NewType('Back', None)
# PersistOpt = str
# ResetOpt = str
ActionList = List[Action]
class State:
def __init__(self, kb: Keyboard, env: OptionSet):
pass
def __del__(self):
pass
@property
def flags(self) -> int:
pass
@property
def context(self) -> Context:
pass
@property
def environment(self) -> OptionSet:
pass
@property
def options(self) -> OptionSet:
pass
def indentify_option_src(opt: Option):
pass
def process_event(vk: VirtualKey,
modifier_state,
state: State,
acts: ActionList):
pass

View file

@ -12,7 +12,7 @@
#include <iomanip>
#include <limits>
#include "json.hpp"
#include "jsonpp.hpp"
#if defined(_MSC_VER)

View file

@ -6,7 +6,7 @@
History: 7 Oct 2018 - TSE - Refactored out of km_kbp_keyboard_api.cpp
*/
#include "keyboard.hpp"
#include "json.hpp"
#include "jsonpp.hpp"
using namespace km::kbp;

View file

@ -15,7 +15,7 @@
#include <keyman/keyboardprocessor.h>
#include "context.hpp"
#include "json.hpp"
#include "jsonpp.hpp"
#include "utfcodec.hpp"
namespace {

View file

@ -14,7 +14,7 @@
#include <keyman/keyboardprocessor.h>
#include "processor.hpp"
#include "json.hpp"
#include "jsonpp.hpp"
#include "state.hpp"

View file

@ -12,11 +12,42 @@
#include "processor.hpp"
#include "state.hpp"
km_kbp_status
km_kbp_event(
km_kbp_state *state,
uint32_t event,
void* data
) {
assert(state != nullptr);
if(state == nullptr) {
return KM_KBP_STATUS_INVALID_ARGUMENT;
}
// event: KM_KBP_EVENT_KEYBOARD_ACTIVATED; data should be nullptr
// future event: KM_KBP_EVENT_KEYBOARD_DEACTIVATED
switch(event) {
case KM_KBP_EVENT_KEYBOARD_ACTIVATED:
assert(data == nullptr);
if(data == nullptr) {
return KM_KBP_STATUS_INVALID_ARGUMENT;
}
break;
default:
return KM_KBP_STATUS_INVALID_ARGUMENT;
}
return state->processor().external_event(state, event, data);
}
km_kbp_status
km_kbp_process_event(km_kbp_state *state,
km_kbp_virtual_key vk,
uint16_t modifier_state,
uint8_t is_key_down) {
assert(state != nullptr);
if(state == nullptr) {
return KM_KBP_STATUS_INVALID_ARGUMENT;
}
return state->processor().process_event(state, vk, modifier_state, is_key_down);
}
@ -24,11 +55,19 @@ km_kbp_status
km_kbp_process_queued_actions(
km_kbp_state *state
) {
assert(state != nullptr);
if(state == nullptr) {
return KM_KBP_STATUS_INVALID_ARGUMENT;
}
return state->processor().process_queued_actions(state);
}
km_kbp_attr const *
km_kbp_get_engine_attrs(km_kbp_state const *state)
{
assert(state != nullptr);
if(state == nullptr) {
return nullptr;
}
return &state->processor().attributes();
}

View file

@ -13,7 +13,7 @@
#include <sstream>
#include <keyman/keyboardprocessor.h>
#include "json.hpp"
#include "jsonpp.hpp"
#include "processor.hpp"
#include "state.hpp"

View file

@ -68,7 +68,7 @@ lib = library('kmnkbp0',
'km_kbp_state_api.cpp',
'km_kbp_debug_api.cpp',
'km_kbp_processevent_api.cpp',
'json.cpp',
'jsonpp.cpp',
'mock/mock_processor.cpp',
'kmx/kmx_consts.cpp',
'kmx/kmx_processevent.cpp',

View file

@ -13,7 +13,7 @@
#include <type_traits>
#include <keyman/keyboardprocessor.h>
#include "json.hpp"
#include "jsonpp.hpp"
#include "utfcodec.hpp"
// Forward declarations

View file

@ -44,6 +44,15 @@ namespace kbp
uint8_t is_key_down
) = 0;
virtual km_kbp_status
external_event(
km_kbp_state* _kmn_unused(state),
uint32_t _kmn_unused(event),
void* _kmn_unused(data)
) {
return KM_KBP_STATUS_OK;
}
virtual km_kbp_attr const & attributes() const = 0;
virtual km_kbp_status validate() const = 0;

View file

@ -14,7 +14,7 @@
#include <iostream>
#include <fstream>
#include <json.hpp>
#include <jsonpp.hpp>
#ifdef __EMSCRIPTEN__
#include <emscripten.h>

View file

@ -13,7 +13,7 @@ endif
e = executable('jsontest', 'jsontest.cpp',
include_directories: [libsrc],
link_args: links + tests_flags,
objects: lib.extract_objects('json.cpp'))
objects: lib.extract_objects('jsonpp.cpp'))
test('jsontest', e, args: 'jsontest.json')
test('jsontestOutput', python, is_parallel: false, args:
cmpfiles + ['jsontest.json', join_paths(stnds, 'jsontest.json')])

View file

@ -1,2 +1,7 @@
[binaries]
c = ['$EMSCRIPTEN_BASE/emcc.py', '-s', 'WASM=1', '-O2']
cpp = ['$EMSCRIPTEN_BASE/em++.py', '-s', 'WASM=1','-O2']
ar = ['$EMSCRIPTEN_BASE/emar.py']
[properties]
root = '$EMSCRIPTEN_BASE/system'

View file

@ -194,7 +194,7 @@ const
CWARN_VisualKeyboardFileMissing = $2097;
CWARN_ExtendedShiftFlagsNotSupportedInKeymanWeb = $2098; // I4118
CWARN_TouchLayoutUnidentifiedKey = $2099; // I4142
CWARN_UnreachableKeyCode = $209A; // I4141
CHINT_UnreachableKeyCode = $109A; // I4141
CWARN_CouldNotCopyJsonFile = $209B; // I4688
CWARN_PlatformNotInTargets = $209C;

View file

@ -25,7 +25,7 @@ BOOL CheckForDeprecatedFeatures(PFILE_KEYBOARD fk) {
// Keyman 7
#define TSS_WINDOWSLANGUAGES 29
*/
int currentLineBackup = currentLine;
int oldCurrentLine = currentLine;
DWORD i;
PFILE_STORE sp;
@ -46,7 +46,7 @@ BOOL CheckForDeprecatedFeatures(PFILE_KEYBOARD fk) {
}
}
currentLine = currentLineBackup;
currentLine = oldCurrentLine;
return TRUE;
}

View file

@ -24,6 +24,8 @@ DWORD VerifyUnreachableRules(PFILE_GROUP gp) {
PFILE_KEY kp = gp->dpKeyArray;
DWORD i;
int oldCurrentLine = currentLine;
std::unordered_map<std::wstring, FILE_KEY> map;
std::unordered_set<int> reportedLines;
@ -43,5 +45,7 @@ DWORD VerifyUnreachableRules(PFILE_GROUP gp) {
}
}
currentLine = oldCurrentLine;
return CERR_None;
}

View file

@ -191,7 +191,7 @@
#define CWARN_VisualKeyboardFileMissing 0x00002097
#define CWARN_ExtendedShiftFlagsNotSupportedInKeymanWeb 0x00002098 // I4118
#define CWARN_TouchLayoutUnidentifiedKey 0x00002099
#define CWARN_UnreachableKeyCode 0x0000209A
#define CHINT_UnreachableKeyCode 0x0000109A
#define CWARN_CouldNotCopyJsonFile 0x0000209B
#define CWARN_PlatformNotInTargets 0x0000209C

View file

@ -53,7 +53,6 @@ uses
CompileErrorCodes in '..\common\delphi\compiler\CompileErrorCodes.pas',
TouchLayout in '..\tike\oskbuilder\TouchLayout.pas',
TouchLayoutDefinitions in '..\tike\oskbuilder\TouchLayoutDefinitions.pas',
TouchLayoutUtils in '..\tike\oskbuilder\TouchLayoutUtils.pas',
KeyboardFonts in '..\common\delphi\general\KeyboardFonts.pas',
KeyboardParser in '..\tike\main\KeyboardParser.pas',
WindowsLanguages in '..\common\delphi\general\WindowsLanguages.pas',

View file

@ -202,7 +202,6 @@
<DCCReference Include="..\common\delphi\compiler\CompileErrorCodes.pas"/>
<DCCReference Include="..\tike\oskbuilder\TouchLayout.pas"/>
<DCCReference Include="..\tike\oskbuilder\TouchLayoutDefinitions.pas"/>
<DCCReference Include="..\tike\oskbuilder\TouchLayoutUtils.pas"/>
<DCCReference Include="..\common\delphi\general\KeyboardFonts.pas"/>
<DCCReference Include="..\tike\main\KeyboardParser.pas"/>
<DCCReference Include="..\common\delphi\general\WindowsLanguages.pas"/>

View file

@ -73,7 +73,6 @@ uses
OnScreenKeyboardData in '..\..\..\common\windows\delphi\visualkeyboard\OnScreenKeyboardData.pas',
TouchLayout in '..\TIKE\oskbuilder\TouchLayout.pas',
TouchLayoutDefinitions in '..\TIKE\oskbuilder\TouchLayoutDefinitions.pas',
TouchLayoutUtils in '..\TIKE\oskbuilder\TouchLayoutUtils.pas',
KeyboardFonts in '..\common\delphi\general\KeyboardFonts.pas',
Keyman.System.Util.RenderLanguageIcon in '..\..\..\common\windows\delphi\ui\Keyman.System.Util.RenderLanguageIcon.pas',
utilicon in '..\..\..\common\windows\delphi\general\utilicon.pas',

View file

@ -179,7 +179,6 @@
<DCCReference Include="..\..\..\common\windows\delphi\visualkeyboard\OnScreenKeyboardData.pas"/>
<DCCReference Include="..\TIKE\oskbuilder\TouchLayout.pas"/>
<DCCReference Include="..\TIKE\oskbuilder\TouchLayoutDefinitions.pas"/>
<DCCReference Include="..\TIKE\oskbuilder\TouchLayoutUtils.pas"/>
<DCCReference Include="..\common\delphi\general\KeyboardFonts.pas"/>
<DCCReference Include="..\..\..\common\windows\delphi\ui\Keyman.System.Util.RenderLanguageIcon.pas"/>
<DCCReference Include="..\..\..\common\windows\delphi\general\utilicon.pas"/>

View file

@ -52,7 +52,7 @@
"chai": "^4.3.4",
"chalk": "^2.4.2",
"jszip": "^3.7.0",
"mocha": "^8.4.0",
"mocha": "^10.0.0",
"ts-node": "^9.1.1"
},
"mocha": {

View file

@ -34,7 +34,7 @@
"@types/ws": "^8.2.2",
"chai": "^4.3.4",
"copyfiles": "^2.4.1",
"mocha": "^9.1.4",
"mocha": "^10.0.0",
"ts-node": "^10.4.0",
"tsc-watch": "^4.5.0",
"typescript": "^4.5.4"

View file

@ -72,14 +72,8 @@ uses
WindowsLanguages in '..\..\..\common\delphi\general\WindowsLanguages.pas',
KeymanWebKeyCodes in '..\..\..\tike\compile\KeymanWebKeyCodes.pas',
kmxfileutils in '..\..\..\..\..\common\windows\delphi\keyboards\kmxfileutils.pas',
TikeUnicodeData in '..\..\..\tike\main\TikeUnicodeData.pas',
UnicodeData in '..\..\..\..\..\common\windows\delphi\charmap\UnicodeData.pas',
ADODB_TLB in '..\..\..\..\..\common\windows\delphi\tlb\ADODB_TLB.pas',
ADOX_TLB in '..\..\..\..\..\common\windows\delphi\tlb\ADOX_TLB.pas',
ttinfo in '..\..\..\..\..\common\windows\delphi\general\ttinfo.pas',
TouchLayoutDefinitions in '..\..\..\tike\oskbuilder\TouchLayoutDefinitions.pas',
TouchLayout in '..\..\..\tike\oskbuilder\TouchLayout.pas',
TouchLayoutUtils in '..\..\..\tike\oskbuilder\TouchLayoutUtils.pas',
KeyboardFonts in '..\..\..\common\delphi\general\KeyboardFonts.pas',
Keyman.Developer.System.Project.kpsProjectFile in '..\..\..\tike\project\Keyman.Developer.System.Project.kpsProjectFile.pas',
Keyman.Developer.System.Project.kpsProjectFileAction in '..\..\..\tike\project\Keyman.Developer.System.Project.kpsProjectFileAction.pas',

View file

@ -153,14 +153,8 @@
<DCCReference Include="..\..\..\common\delphi\general\WindowsLanguages.pas"/>
<DCCReference Include="..\..\..\tike\compile\KeymanWebKeyCodes.pas"/>
<DCCReference Include="..\..\..\..\..\common\windows\delphi\keyboards\kmxfileutils.pas"/>
<DCCReference Include="..\..\..\tike\main\TikeUnicodeData.pas"/>
<DCCReference Include="..\..\..\..\..\common\windows\delphi\charmap\UnicodeData.pas"/>
<DCCReference Include="..\..\..\..\..\common\windows\delphi\tlb\ADODB_TLB.pas"/>
<DCCReference Include="..\..\..\..\..\common\windows\delphi\tlb\ADOX_TLB.pas"/>
<DCCReference Include="..\..\..\..\..\common\windows\delphi\general\ttinfo.pas"/>
<DCCReference Include="..\..\..\tike\oskbuilder\TouchLayoutDefinitions.pas"/>
<DCCReference Include="..\..\..\tike\oskbuilder\TouchLayout.pas"/>
<DCCReference Include="..\..\..\tike\oskbuilder\TouchLayoutUtils.pas"/>
<DCCReference Include="..\..\..\common\delphi\general\KeyboardFonts.pas"/>
<DCCReference Include="..\..\..\tike\project\Keyman.Developer.System.Project.kpsProjectFile.pas"/>
<DCCReference Include="..\..\..\tike\project\Keyman.Developer.System.Project.kpsProjectFileAction.pas"/>

View file

@ -36,6 +36,8 @@ implementation
uses
System.SysUtils,
Winapi.ActiveX,
compile,
CompilePackage,
Keyman.Developer.System.Project.kmnProjectFileAction,
@ -154,5 +156,8 @@ begin
end;
initialization
CoInitializeEx(nil, COINIT_APARTMENTTHREADED);
TDUnitX.RegisterTestFixture(TCompilePackageVersioningTest);
finalization
CoUninitialize;
end.

View file

@ -102,6 +102,7 @@ uses
Winapi.Windows,
System.Character,
System.Classes,
System.Generics.Collections,
System.UITypes,
compile,
@ -214,6 +215,7 @@ type
FTouchLayoutFont: string;
FFix183_LadderLength: Integer;
FCloseBrace: Boolean; // I4872
FUnreachableKeys: TList<PFILE_KEY>;
function JavaScript_String(ch: DWord): string; // I2242
@ -254,6 +256,8 @@ type
function IsKeyboardVersion15OrLater: Boolean;
function WriteBeginStatement(const name: string;
groupIndex: Integer): string;
function FormatKeyForErrorMessage(fkp: PFILE_KEY;
FMnemonic: Boolean): string;
public
function Compile(AOwnerProject: TProject; const InFile: string; const OutFile: string; Debug: Boolean; Callback: TCompilerCallbackW): Boolean; // I3681 // I4140 // I4688 // I4866
constructor Create;
@ -273,14 +277,11 @@ uses
CompileErrorCodes,
JsonUtil,
KeymanDeveloperOptions,
KeyboardParser,
Keyman.System.KeyboardUtils,
KeymanWebKeyCodes,
kmxfileutils,
TouchLayout,
TouchLayoutUtils,
Unicode,
UnicodeData,
utilstr,
VisualKeyboard,
VKeys;
@ -303,6 +304,8 @@ var
WarnDeprecatedCode: Boolean;
Data: string;
begin
FUnreachableKeys.Clear;
FCallback := Callback;
FInFile := InFile;
FOutFile := OutFile; // I4140 // I4155 // I4154
@ -375,6 +378,7 @@ end;
constructor TCompileKeymanWeb.Create;
begin
FUnreachableKeys := TList<PFILE_KEY>.Create;
FillChar(fk, sizeof(fk), 0);
FFix183_LadderLength := FKeymanDeveloperOptions.Fix183_LadderLength; // How frequently to break ladders
end;
@ -382,6 +386,7 @@ end;
destructor TCompileKeymanWeb.Destroy;
begin
// TODO: Free FK values
FUnreachableKeys.Free;
inherited;
end;
@ -800,6 +805,63 @@ const // I1585 - add space to conversion
VK_NUMLOCK, // &H90
VK_SCROLL); // &H91
function TCompileKeymanWeb.FormatKeyForErrorMessage(fkp: PFILE_KEY; FMnemonic: Boolean): string;
function FormatShift(ShiftFlags: DWord): string;
const
mask: array[0..13] of string = (
'LCTRL', // 0X0001
'RCTRL', // 0X0002
'LALT', // 0X0004
'RALT', // 0X0008
'SHIFT', // 0X0010
'CTRL', // 0X0020
'ALT', // 0X0040
'???', // Reserved
'CAPS', // 0X0100
'NCAPS', // 0X0200
'NUMLOCK', // 0X0400
'NNUMLOCK', // 0X0800
'SCROLLLOCK', // 0X1000
'NSCROLLLOCK' // 0X2000
);
var
i: Integer;
begin
Result := '';
for i := 0 to High(mask) do
begin
if ShiftFlags and (1 shl i) <> 0 then
begin
Result := Result + mask[i] + ' ';
end;
end;
end;
begin
if not FMnemonic then
begin
if (fkp.ShiftFlags and KMX_ISVIRTUALKEY) = KMX_ISVIRTUALKEY then
begin
if Ord(fkp.Key) < 256
then Result := Format('[%s%s]', [FormatShift(fkp.ShiftFlags), VKeyNames[Ord(fkp.Key)]])
else Result := Format('[%sK_%x]', [FormatShift(fkp.ShiftFlags), Ord(fkp.Key)]);
end
else
begin
Result := Format('''%s''', [fkp.Key]);
end;
end
else
begin
if (fkp.ShiftFlags and KMX_VIRTUALCHARKEY) = KMX_VIRTUALCHARKEY
then Result := Format('[%s''%s'']', [FormatShift(fkp.ShiftFlags), fkp.Key])
else Result := Format('''%s''', [fkp.Key]);
end;
end;
function TCompileKeymanWeb.JavaScript_Key(fkp: PFILE_KEY; FMnemonic: Boolean): Integer;
var
@ -835,7 +897,13 @@ begin
if (Result = 0) or (Result >= Ord(Low(TKeymanWebTouchStandardKey))) then // I4141
begin
ReportError(fkp.Line, CWARN_UnreachableKeyCode, 'The rule will never be matched because its key code is never fired.');
if not FUnreachableKeys.Contains(fkp) then
begin
ReportError(fkp.Line, CHINT_UnreachableKeyCode,
'The rule will never be matched for key '+
FormatKeyForErrorMessage(fkp,FMnemonic)+' because its key code is never fired.');
FUnreachableKeys.Add(fkp);
end;
end;
end;

View file

@ -613,6 +613,11 @@ body:not(.text-controls-in-toolbar) input#inpSubKeyCap {
color: white;
}
#kbd.desktop .key-size {
/* the layout is fixed on desktop so key size is not useful */
display: none;
}
/* Position flick keys relative to the flick grid, by hand */
#flick .key {

View file

@ -12,10 +12,7 @@ $(function() {
this.getPresentation = function () {
var platform = $('#selPlatformPresentation').val();
//if(platform == 'tablet') return 'tablet-ipad';
//if(platform == 'phone') return 'phone-iphone5';
return platform;
return $('#selPlatformPresentation').val();
}
this.saveSelection = function() {

View file

@ -604,7 +604,8 @@ $(function() {
"tablet-ipad-landscape": { "x": 829, "y": 299, "name": "iPad (landscape)" }, // 829x622 = iPad tablet box size; (97,101)-(926,723)
"tablet-ipad-portrait": { "x": 605, "y": 300, "name": "iPad (portrait)" }, // 605x806 = iPad tablet box size; (98,94)-(703,900)
"phone-iphone5-landscape": { "x": 731, "y": 196, "name": "iPhone 5 (landscape)" }, // 731x412 = iPhone box size; (144,39)-(875,451)
"phone-iphone5-portrait": { "x": 526, "y": 266, "name": "iPhone 5 (portrait)"} // 528x936 = iPhone box size; (90,204)-(618,1040)
"phone-iphone5-portrait": { "x": 526, "y": 266, "name": "iPhone 5 (portrait)"}, // 528x936 = iPhone box size; (90,204)-(618,1040)
"desktop": { "x": 640, "y": 300, "name": "Desktop" },
};
this.keyMargin = 15;

View file

@ -3,32 +3,31 @@
#
# @darcywong00 @mcdurdin @ermshiperete @rc-swag @SabineSIL @sgschantz
/android/ @darcywong00 @rc-swag
/android/ @darcywong00 @mcdurdin
/common/ @mcdurdin @jahorton
/core/ @mcdurdin @jahorton
/common/ @mcdurdin @rc-swag
/core/ @mcdurdin @rc-swag
/common/lexical-model-types/ @jahorton @mcdurdin
/common/models/ @jahorton @mcdurdin
/common/predictive-text/ @jahorton @mcdurdin
/common/schemas/ @mcdurdin @jahorton
/common/test/ @mcdurdin @ermshiperete
/common/web/ @jahorton @ermshiperete @mcdurdin
/common/web/ @jahorton @sgschantz @mcdurdin
/developer/ @mcdurdin @darcywong00
/docs/ @mcdurdin @jahorton
/ios/ @sgschantz @mcdurdin
/ios/ @sgschantz @jahorton
/linux/ @ermshiperete @darcywong00
/mac/ @sgschantz @SabineSIL
/oem/firstvoices/android/ @darcywong00 @rc-swag
/oem/firstvoices/common/ @mcdurdin @jahorton
/oem/firstvoices/ios/ @sgschantz @mcdurdin
/oem/firstvoices/windows/ @rc-swag @sgschantz
/oem/firstvoices/android/ @darcywong00 @mcdurdin
/oem/firstvoices/common/ @mcdurdin @rc-swag
/oem/firstvoices/ios/ @sgschantz @jahorton
/oem/firstvoices/windows/ @rc-swag @ermshiperete
/resources/ @mcdurdin @jahorton
# Web is currently shared between Marc and Joshua:
/web/ @jahorton @ermshiperete @mcdurdin
/web/ @jahorton @sgschantz @mcdurdin
/windows/ @rc-swag @ermshiperete
/windows/ @rc-swag @sgschantz
# Override for windows/src:
/windows/src/developer/ @mcdurdin @darcywong00

View file

@ -64,3 +64,23 @@ All dependencies are already installed if you followed the instructions under [P
Building:
* [Building Keyman Core](../../core/doc/BUILDING.md)
## Docker Builder
The Docker builder allows you to perform a linux build from anywhere Docker is supported.
To build the docker image:
```shell
cd ../../linux
docker pull ubuntu:latest
docker build . -t keymanapp/keyman-linux-builder:latest
```
Once the image is built, it may be used to build parts of Keyman.
```shell
cd ../core
# keep linux build artifacts separate
mkdir -p build/linux
docker run -it --rm -v $(pwd)/..:/home/build -v $(pwd)/build/linux:/home/build/core/build keyman-linux-builder:latest bash -c 'cd core; bash build.sh -d'
```

View file

@ -30,7 +30,7 @@ public enum KeymanHosts {
case .alpha:
fallthrough
case .beta:
return URL.init(string: "https://api.keyman-staging.com")!
return URL.init(string: "https://api.keyman.com")! // #7227 disabling: "https://api.keyman-staging.com")!
case .stable:
return URL.init(string: "https://api.keyman.com")!
}
@ -54,7 +54,7 @@ public enum KeymanHosts {
case .alpha:
fallthrough
case .beta:
return URL.init(string: "https://help.keyman-staging.com")!
return URL.init(string: "https://help.keyman.com")! // #7227 disabling: "https://help.keyman-staging.com")!
case .stable:
return URL.init(string: "https://help.keyman.com")!
}
@ -78,7 +78,7 @@ public enum KeymanHosts {
case .alpha:
fallthrough
case .beta:
return URL.init(string: "https://keyman-staging.com")!
return URL.init(string: "https://keyman.com")! // #7227 disabling: "https://keyman-staging.com")!
case .stable:
return URL.init(string: "https://keyman.com")!
}

Some files were not shown because too many files have changed in this diff Show more