diff --git a/HISTORY.md b/HISTORY.md
index b5d6ea2caa..89517888c2 100644
--- a/HISTORY.md
+++ b/HISTORY.md
@@ -1,5 +1,37 @@
# Keyman Version History
+## 18.0.57 alpha 2024-06-17
+
+* fix(web): fix id of longpress keys with modifier set in touch layout (#11783)
+* fix(web): prevent desktop OSK crash when addKeyboards is called before engine init (#11786)
+* fix(core): serialize tests for core/wasm on mac agents (#11795)
+* fix(developer): refactor kmcmplib compiler messages to use map (#11738)
+* fix(developer): make native compilation of kmcmplib under Linux possible (#11779)
+* feat(core): devolve regex to javascript (#11777)
+* feat(core): remove ICU from core under wasm (#11778)
+* fix(linux): restart ibus after manual integration test run (#11775)
+
+## 18.0.56 alpha 2024-06-14
+
+* feat(core): devolve normalization to js (#11541)
+* fix(developer): show message if no more platforms to add to touch layout editor (#11759)
+* docs(developer): context help in package-editor and put the existing context help in their own tab comments (#11760)
+* docs(developer): context help in keyboard-editor section (#11754)
+* docs(developer): context help in new-project section (#11767)
+* docs(developer): context help for new-project-parameters in keyman developer (#11769)
+* docs(developer): context help for Select BCP 47 tag in Keyman Developer (#11770)
+* change(common): update esbuild to 0.18.9 (#11693)
+* change(web): more prep for better async prediction handling (#10347)
+* fix(web): set new-context rules' device to match that of the active OSK (#11743)
+* chore(linux): Update debian changelog (#11671)
+* fix(web): add limited Array.from polyfill for lm-worker use (#11732)
+
+## 18.0.55 alpha 2024-06-13
+
+* fix(developer): handle missing OSK when importing a Windows keyboard into a touch-only project (#11720)
+* fix(developer): verify email addresses in .kps and .keyboard_info (#11735)
+* change(web): prep for better asynchronous prediction handling (#10343)
+
## 18.0.54 alpha 2024-06-12
* fix(common): remove subpackage entries for older TS version (#11745)
diff --git a/VERSION.md b/VERSION.md
index 1d6c319a44..43a4d909f5 100644
--- a/VERSION.md
+++ b/VERSION.md
@@ -1 +1 @@
-18.0.55
\ No newline at end of file
+18.0.58
\ No newline at end of file
diff --git a/common/models/wordbreakers/src/default/index.ts b/common/models/wordbreakers/src/default/index.ts
index 20e159ed19..ef50ab4b31 100644
--- a/common/models/wordbreakers/src/default/index.ts
+++ b/common/models/wordbreakers/src/default/index.ts
@@ -274,7 +274,7 @@ export class BreakerContext {
* @param chunk a chunk of text. Starts and ends at word boundaries.
*/
function isNonSpace(chunk: string, options?: DefaultWordBreakerOptions): boolean {
- return !Array.from(chunk).map((char) => property(char, options)).every(wb => (
+ return !chunk.split('').map((char) => property(char, options)).every(wb => (
wb === WordBreakProperty.CR ||
wb === WordBreakProperty.LF ||
wb === WordBreakProperty.Newline ||
diff --git a/common/web/lm-worker/build-polyfiller.js b/common/web/lm-worker/build-polyfiller.js
index 20f3ab587f..1978038836 100644
--- a/common/web/lm-worker/build-polyfiller.js
+++ b/common/web/lm-worker/build-polyfiller.js
@@ -144,6 +144,9 @@ let sourceFileSet = [
// Needed for Android / Chromium browser pre-41.
loadPolyfill('../../../node_modules/string.prototype.codepointat/codepointat.js', 'src/polyfills/string.codepointat.js'),
// Needed for Android / Chromium browser pre-45.
+ // Not used in this codebase, but used by some compiled model defaults.
+ loadPolyfill('src/polyfills/array.from.js', 'src/polyfills/array.from.js'),
+ // Needed for Android / Chromium browser pre-45.
loadPolyfill('src/polyfills/array.fill.js', 'src/polyfills/array.fill.js'),
// Needed for Android / Chromium browser pre-45.
loadPolyfill('src/polyfills/array.findIndex.js', 'src/polyfills/array.findIndex.js'),
diff --git a/common/web/lm-worker/src/polyfills/array.from.js b/common/web/lm-worker/src/polyfills/array.from.js
new file mode 100644
index 0000000000..ae6ba84600
--- /dev/null
+++ b/common/web/lm-worker/src/polyfills/array.from.js
@@ -0,0 +1,44 @@
+(function() {
+ if(!Array.from) {
+ function isHighSurrogate(codeUnit) {
+ codeUnit = codeUnit.charCodeAt(0);
+ return codeUnit >= 0xD800 && codeUnit <= 0xDBFF;
+ }
+
+ function isLowSurrogate(codeUnit) {
+ codeUnit = codeUnit.charCodeAt(0);
+ return codeUnit >= 0xDC00 && codeUnit <= 0xDFFF;
+ }
+
+ Array.from = function (obj) {
+ if(Array.isArray(obj)) {
+ // Simple array clone
+ return obj.slice();
+ } else if(typeof obj == 'string') {
+ // Array.from is surrogate-aware and will not split surrogate pairs.
+ // We can start with a full split and then remerge the pairs.
+ var simpleSplit = obj.split('');
+
+ /** @type {string[]} */
+ var finalSplit = [];
+ /** @type {number} */
+ var i;
+
+ while(simpleSplit.length > 0) {
+ // Do we have a surrogate pair?
+ var a = simpleSplit.shift();
+ if(isHighSurrogate(a) && (isLowSurrogate(simpleSplit[0] || ''))) {
+ // yes, so merge them before pushing.
+ a = a + simpleSplit.shift();
+ console.log(a);
+ } // else: 'no', so just push the current char to the array and continue
+
+ finalSplit.push(a);
+ }
+ return finalSplit;
+ } else {
+ throw "Unexpected + nonpolyfilled use of Array.from encountered; aborting";
+ }
+ }
+ }
+}());
\ No newline at end of file
diff --git a/core/commands.inc.sh b/core/commands.inc.sh
index 9f1365cdda..f2c49d88ca 100644
--- a/core/commands.inc.sh
+++ b/core/commands.inc.sh
@@ -91,7 +91,14 @@ do_test() {
if [[ $target =~ ^(x86|x64)$ ]]; then
cmd //C build.bat $target $BUILDER_CONFIGURATION test $testparams
else
- meson test -C "$MESON_PATH" $testparams
+ if [[ $target == wasm ]] && [[ $BUILDER_OS == mac ]]; then
+ # 11794 -- parallel tests failing on some mac build agents; temporary
+ # mitigation until we diagnose root cause
+ meson test -j 1 -C "$MESON_PATH" $testparams
+ else
+ meson test -C "$MESON_PATH" $testparams
+ fi
+
fi
builder_finish_action success test:$target
}
diff --git a/core/src/actions_normalize.cpp b/core/src/actions_normalize.cpp
index 3384fda975..78bada7b6f 100644
--- a/core/src/actions_normalize.cpp
+++ b/core/src/actions_normalize.cpp
@@ -15,12 +15,12 @@
#include "state.hpp"
#include "option.hpp"
#include "debuglog.h"
-#include "core_icu.h"
+#include "util_normalize.hpp"
+#include "utfcodec.hpp"
+#include "kmx/kmx_xstring.h"
-// forward declarations
-
-icu::UnicodeString context_items_to_unicode_string(km::core::context const *context);
-km_core_usv *unicode_string_to_usv(icu::UnicodeString& src);
+// forward declaration
+bool context_items_to_unicode_string(km::core::context const *context, std::u32string &str);
/**
* Normalize the output from an action to NFC, across the context | output
@@ -65,32 +65,12 @@ bool km::core::actions_normalize(
cached_context.
*/
- /*
- Initialization
- */
-
- UErrorCode icu_status = U_ZERO_ERROR;
- const icu::Normalizer2 *nfc = icu::Normalizer2::getNFCInstance(icu_status);
- assert(U_SUCCESS(icu_status));
- if(!U_SUCCESS(icu_status)) {
- DebugLog("getNFCInstance failed with %x", icu_status);
+ std::u32string output(actions.output);
+ std::u32string cached_context_string, app_context_string;
+ if (!context_items_to_unicode_string(cached_context, cached_context_string)) {
return false;
}
-
- const icu::Normalizer2 *nfd = icu::Normalizer2::getNFDInstance(icu_status);
- assert(U_SUCCESS(icu_status));
- if(!U_SUCCESS(icu_status)) {
- DebugLog("getNFDInstance failed with %x", icu_status);
- return false;
- }
-
- icu::UnicodeString output = icu::UnicodeString::fromUTF32(reinterpret_cast(actions.output), -1);
- icu::UnicodeString cached_context_string = context_items_to_unicode_string(cached_context);
- icu::UnicodeString app_context_string = context_items_to_unicode_string(app_context);
- assert(!output.isBogus());
- assert(!cached_context_string.isBogus());
- assert(!app_context_string.isBogus());
- if(output.isBogus() || cached_context_string.isBogus() || app_context_string.isBogus()) {
+ if (!context_items_to_unicode_string(app_context, app_context_string)) {
return false;
}
int nfu_to_delete = 0;
@@ -98,20 +78,25 @@ bool km::core::actions_normalize(
/*
Further debug assertion of inputs
*/
-
- assert(nfd->isNormalized(output, icu_status) && U_SUCCESS(icu_status));
- assert(nfd->isNormalized(cached_context_string, icu_status) && U_SUCCESS(icu_status));
+ assert(km::core::util::is_nfd(output));
+ assert(km::core::util::is_nfd(cached_context_string));
/*
The keyboard processor will have updated the cached_context already,
- applying the transform to it, so we need to rewind this. Remove the output
- from cached_context_string to start
+ applying the transform to it, so we need to rewind this.
+
+ Assert that 'cached_context_string' ends with 'output'
+
+ Remove the output
+ from cached_context_string to start.
*/
assert(cached_context_string.length() >= output.length());
- int n = cached_context_string.length() - output.length();
+ size_t n = cached_context_string.length() - output.length();
+ // auto end_cached = cached_context_string.substr(n, output.length());
+ // assert(end_cached == output);
assert(cached_context_string.compare(n, output.length(), output) == 0);
- cached_context_string.remove(n);
+ cached_context_string.resize(n);
/*
While cached_context is guaranteed to be normalized, actions->output may not
@@ -119,21 +104,22 @@ bool km::core::actions_normalize(
normalization in our output, we now need to look for a normalization
boundary prior to the intersection of the cached_context and the output.
*/
- if(!output.isEmpty()) {
- while(n > 0 && !nfd->hasBoundaryBefore(output[0])) {
+ if(!output.empty()) {
+ while(n > 0 && !km::core::util::has_nfd_boundary_before(output[0])) {
// The output may interact with the context further in normalization. We
// need to copy characters back further until we reach a normalization
// boundary.
// Remove last code point from the context ...
- n = cached_context_string.moveIndex32(n, -1);
- UChar32 chr = cached_context_string.char32At(n);
- cached_context_string.remove(n);
+ auto len = cached_context_string.length();
+ assert(len>0);
+ auto chr = cached_context_string.at(len-1);
+ cached_context_string.resize(len-1);
// And prepend it to the output ...
- output.insert(0, chr);
+ output.insert(0, 1, chr);
}
}
@@ -148,25 +134,20 @@ bool km::core::actions_normalize(
its normalized form matches the cached_context normalized form.
*/
- while(app_context_string.countChar32()) {
- icu::UnicodeString app_context_nfd;
- nfd->normalize(app_context_string, app_context_nfd, icu_status);
- assert(U_SUCCESS(icu_status));
- if(!U_SUCCESS(icu_status)) {
- DebugLog("nfd->normalize failed with %x", icu_status);
+ while(!app_context_string.empty()) {
+ auto app_context_nfd = app_context_string;
+ if(!km::core::util::normalize_nfd(app_context_nfd)) {
+ DebugLog("nfd->normalize failed");
return false;
}
- if(app_context_nfd.compare(cached_context_string) == 0) {
+ if(app_context_nfd == cached_context_string) {
break;
}
- // remove the last UChar32
- int32_t lastUChar32 = app_context_string.length()-1;
- // adjust pointer to get the entire char (i.e. so we don't slice a non-BMP char)
- lastUChar32 = app_context_string.getChar32Start(lastUChar32);
- // remove the UChar32 (1 or 2 code units)
- app_context_string.remove(lastUChar32);
+ size_t len = app_context_string.length();
+ // remove the cp at end
+ app_context_string.resize(len-1);
nfu_to_delete++;
}
@@ -174,17 +155,15 @@ bool km::core::actions_normalize(
Normalize our output string
*/
- icu::UnicodeString output_nfc;
- nfc->normalize(output, output_nfc, icu_status);
- assert(U_SUCCESS(icu_status));
- if(!U_SUCCESS(icu_status)) {
- DebugLog("nfc->normalize failed with %x", icu_status);
+ auto output_nfc = output;
+ if(!km::core::util::normalize_nfc(output_nfc)) {
+ DebugLog("nfc->normalize failed");
return false;
}
- auto new_output = unicode_string_to_usv(output_nfc);
- if(!new_output) {
- // error logging handled in unicode_string_to_usv
+ auto new_output = km::core::util::string_to_usv(output_nfc);
+ assert(new_output != nullptr);
+ if(new_output == nullptr) {
return false;
}
@@ -197,8 +176,8 @@ bool km::core::actions_normalize(
app_context_string.append(output_nfc);
km_core_context_item *app_context_items = nullptr;
km_core_status status = KM_CORE_STATUS_OK;
- if((status = context_items_from_utf16(app_context_string.getTerminatedBuffer(), &app_context_items)) != KM_CORE_STATUS_OK) {
- DebugLog("context_items_from_utf16 failed with %x", status);
+ if((status = context_items_from_utf32(app_context_string.c_str(), &app_context_items)) != KM_CORE_STATUS_OK) {
+ DebugLog("context_items_from_string failed with %x", status);
delete [] new_output;
return false;
}
@@ -224,21 +203,19 @@ bool km::core::actions_normalize(
/**
* Helper to convert km_core_context list into a icu::UnicodeString
*/
-icu::UnicodeString context_items_to_unicode_string(km::core::context const *context) {
- icu::UnicodeString nullString;
- nullString.setToBogus();
+bool context_items_to_unicode_string(km::core::context const *context, std::u32string &str) {
km_core_context_item *items = nullptr;
km_core_status status;
if((status = km_core_context_get(static_cast(context), &items)) != KM_CORE_STATUS_OK) {
DebugLog("Failed to retrieve context with %s", status);
- return nullString;
+ return false;
}
size_t buf_size = 0;
if((status = context_items_to_utf32(items, nullptr, &buf_size)) != KM_CORE_STATUS_OK) {
DebugLog("Failed to retrieve context size with %s", status);
km_core_context_items_dispose(items);
- return nullString;
+ return false;
}
km_core_usv *buf = new km_core_usv[buf_size];
@@ -246,38 +223,15 @@ icu::UnicodeString context_items_to_unicode_string(km::core::context const *cont
DebugLog("Failed to retrieve context with %s", status);
km_core_context_items_dispose(items);
delete [] buf;
- return nullString;
+ return false;
}
- auto result = icu::UnicodeString::fromUTF32(reinterpret_cast(buf), -1);
+ str = std::u32string(buf, buf_size - 1); // don't include terminating null
km_core_context_items_dispose(items);
delete [] buf;
- return result;
+ return true;
}
-/**
- * Helper to convert icu::UnicodeString to a UTF-32 km_core_usv buffer,
- * nul-terminated
- */
-km_core_usv *unicode_string_to_usv(icu::UnicodeString& src) {
- UErrorCode icu_status = U_ZERO_ERROR;
-
- km_core_usv *dst = new km_core_usv[src.length() + 1];
-
- src.toUTF32(reinterpret_cast(dst), src.length(), icu_status);
-
- assert(U_SUCCESS(icu_status));
- if(!U_SUCCESS(icu_status)) {
- DebugLog("toUTF32 failed with %x", icu_status);
- delete[] dst;
- return nullptr;
- }
-
- dst[src.length()] = 0;
- return dst;
-}
-
-
/**
* Refresh app_context to match the cached_context. Does not do normalization,
diff --git a/core/src/context.hpp b/core/src/context.hpp
index 91cc9688cd..f56d637eff 100644
--- a/core/src/context.hpp
+++ b/core/src/context.hpp
@@ -87,6 +87,30 @@ km_core_status
context_items_from_utf16(km_core_cu const *text,
km_core_context_item **out_ptr);
+/**
+ * Convert a UTF32 encoded Unicode string into an array of `km_core_context_item`
+ * structures. Allocates memory as needed.
+ *
+ * @return km_core_status
+ * * `KM_CORE_STATUS_OK`: On success.
+ * * `KM_CORE_STATUS_INVALID_ARGUMENT`: If non-optional parameters are
+ * null.
+ * * `KM_CORE_STATUS_NO_MEM`: In the event not enough memory can be
+ * allocated for the output buffer.
+ * * `KM_CORE_STATUS_INVALID_UTF`: In the event the UTF32 string cannot
+ * be decoded.
+ *
+ * @param text a pointer to a null terminated array of utf32 encoded data.
+ * @param out_ptr a pointer to the result variable: A pointer to the start of
+ * the `km_core_context_item` array containing the representation
+ * of the input string. Terminated with a type of
+ * `KM_CORE_CT_END`. Must be disposed of with
+ * `km_core_context_items_dispose`.
+ */
+km_core_status
+context_items_from_utf32(km_core_usv const *text,
+ km_core_context_item **out_ptr);
+
/**
* Convert a context item array into a UTF-16 encoded string placing it into the
* supplied buffer of specified size, and return the number of code units
diff --git a/core/src/core_icu.cpp b/core/src/core_icu.cpp
new file mode 100644
index 0000000000..cfaac47c50
--- /dev/null
+++ b/core/src/core_icu.cpp
@@ -0,0 +1,72 @@
+/*
+ Copyright: Β© SIL International.
+ Description: Common LDML utilities
+ Create Date: 24 May 2024
+ Authors: Steven R. Loomis
+*/
+
+#include "core_icu.h"
+
+#if !KMN_NO_ICU
+
+namespace km {
+namespace core {
+namespace util {
+
+/**
+ * Helper to convert icu::UnicodeString to a UTF-32 km_core_usv buffer,
+ * nul-terminated
+ */
+km_core_usv *unicode_string_to_usv(icu::UnicodeString& src) {
+ UErrorCode icu_status = U_ZERO_ERROR;
+
+ km_core_usv *dst = new km_core_usv[src.length() + 1];
+
+ src.toUTF32(reinterpret_cast(dst), src.length(), icu_status);
+
+ if(!UASSERT_SUCCESS(icu_status)) {
+ DebugLog("toUTF32 failed with %x", icu_status);
+ delete[] dst;
+ return nullptr;
+ }
+
+ dst[src.length()] = 0;
+ return dst;
+}
+
+/**
+ * Internal function to normalize with a specified mode.
+ * Note: that this function _does_ assert failure, so it is not
+ * required to assert its return code. The return is provided so
+ * that callers can exit (such as making no change) if there was failure.
+ *
+ * Also note that "failure" here is something catastrophic: ICU not initialized,
+ * or, more likely, some low memory situation. Does not fail on "bad" data.
+ * @param n the ICU Normalizer to use
+ * @param str input/output string
+ * @param status error code, must be initialized on input
+ * @return false if failure
+ */
+bool normalize(const icu::Normalizer2 *n, std::u16string &str, UErrorCode &status) {
+ if(!UASSERT_SUCCESS(status)) {
+ return false;
+ }
+ assert(n != nullptr);
+ icu::UnicodeString dest;
+ icu::UnicodeString src = icu::UnicodeString(str.data(), (int32_t)str.length());
+ n->normalize(src, dest, status);
+ // the next line here will assert
+ if (!UASSERT_SUCCESS(status)) {
+ return false;
+ } else {
+ str.assign(dest.getBuffer(), dest.length());
+ return true;
+ }
+}
+
+
+}
+}
+} /* end km::core::util */
+
+#endif
diff --git a/core/src/core_icu.h b/core/src/core_icu.h
index d95d198637..20b13b0456 100644
--- a/core/src/core_icu.h
+++ b/core/src/core_icu.h
@@ -3,15 +3,42 @@
*/
#pragma once
+#ifdef __EMSCRIPTEN__
+// define this in tests to keep ICU around
+# if !defined(KMN_IN_LDML_TESTS)
+# if !defined(KMN_NO_ICU)
+// under wasm, turn off ICU except in tests.
+# define KMN_NO_ICU 1
+# endif
+# endif
+#elif !defined(KMN_NO_ICU)
+# define KMN_NO_ICU 0
+#endif
+
+#if KMN_NO_ICU
+
+// NO ICU
+
+// any shims needed here for disabling ICU
+
+#else
+
+// YES ICU
+
#if !defined(HAVE_ICU4C)
-#error icu4c is required for this code
+# error icu4c is required for this code
#endif
#define U_FALLTHROUGH
#include "unicode/utypes.h"
#include "unicode/unistr.h"
#include "unicode/normalizer2.h"
+#include "unicode/uniset.h"
+#include "unicode/usetiter.h"
+#include "unicode/regex.h"
+#include "unicode/utext.h"
+#include "keyman_core.h"
#include "debuglog.h"
#include
@@ -29,5 +56,40 @@ inline bool uassert_success(const char *file, int line, const char *function, UE
/**
* Assert an ICU4C UErrorCode
* the first assert is for debug builds, the second triggers the debuglog and has the return value.
+ * @returns true on success
* */
#define UASSERT_SUCCESS(status) (assert(U_SUCCESS(status)), uassert_success(__FILE__, __LINE__, __FUNCTION__, status))
+
+// ------------------ some ICU C++ utilities ----------------------------
+
+namespace km {
+namespace core {
+namespace util {
+
+/**
+ * Convert a UnicodeString to a km_core_usv array
+ * @return the 0-terminated array. Caller owns storage.
+ */
+km_core_usv *unicode_string_to_usv(icu::UnicodeString& src);
+
+/**
+ * Internal function to normalize with a specified mode.
+ * Note: that this function _does_ assert failure, so it is not
+ * required to assert its return code. The return is provided so
+ * that callers can exit (such as making no change) if there was failure.
+ *
+ * Also note that "failure" here is something catastrophic: ICU not initialized,
+ * or, more likely, some low memory situation. Does not fail on "bad" data.
+ * @param n the ICU Normalizer to use
+ * @param str input/output string
+ * @param status error code, must be initialized on input
+ * @return false if failure
+ */
+bool normalize(const icu::Normalizer2 *n, std::u16string &str, UErrorCode &status);
+
+
+}
+}
+}
+
+#endif /* KMN_NO_ICU */
diff --git a/core/src/km_core_context_api.cpp b/core/src/km_core_context_api.cpp
index dd27ab7e5c..594080d8d0 100644
--- a/core/src/km_core_context_api.cpp
+++ b/core/src/km_core_context_api.cpp
@@ -115,6 +115,13 @@ context_items_from_utf16(km_core_cu const *text,
}
+km_core_status
+context_items_from_utf32(km_core_usv const *text,
+ km_core_context_item **out_ptr)
+{
+ return _context_items_from(reinterpret_cast(text), out_ptr);
+}
+
km_core_status context_items_to_utf8(km_core_context_item const *ci,
char *buf, size_t * sz_ptr)
{
diff --git a/core/src/kmx/kmx_xstring.cpp b/core/src/kmx/kmx_xstring.cpp
index a1d5ff518a..794ba26889 100644
--- a/core/src/kmx/kmx_xstring.cpp
+++ b/core/src/kmx/kmx_xstring.cpp
@@ -50,6 +50,15 @@ size_t km::core::kmx::u16len(const km_core_cu *p) {
return i;
}
+size_t km::core::kmx::u32len(const km_core_usv *p) {
+ int i = 0;
+ while (*p) {
+ p++;
+ i++;
+ }
+ return i;
+}
+
int km::core::kmx::u16cmp(const km_core_cu *p, const km_core_cu *q) {
while (*p && *q) {
if (*p != *q) return *p - *q;
@@ -107,6 +116,11 @@ km_core_cu *km::core::kmx::u16dup(km_core_cu *src) {
memcpy(dup, src, (u16len(src) + 1) * sizeof(km_core_cu));
return dup;
}
+km_core_usv *km::core::kmx::u32dup(const km_core_usv *src) {
+ km_core_usv *dup = new km_core_usv[u32len(src) + 1];
+ memcpy(dup, src, (u32len(src) + 1) * sizeof(src[0]));
+ return dup;
+}
/*
* int xstrlen( PKMX_BYTE p );
diff --git a/core/src/kmx/kmx_xstring.h b/core/src/kmx/kmx_xstring.h
index a82e959e56..e17c003091 100644
--- a/core/src/kmx/kmx_xstring.h
+++ b/core/src/kmx/kmx_xstring.h
@@ -117,6 +117,9 @@ int u16ncmp(const km_core_cu *p, const km_core_cu *q, size_t count);
km_core_cu *u16tok(km_core_cu *p, km_core_cu ch, km_core_cu **ctx);
km_core_cu *u16dup(km_core_cu *src);
+size_t u32len(const km_core_usv *p);
+km_core_usv *u32dup(const km_core_usv *src);
+
//KMX_BOOL MapUSCharToVK(KMX_WORD ch, PKMX_WORD puKey, PKMX_DWORD puShiftFlags);
// --- implementation ---
diff --git a/core/src/ldml/ldml_markers.cpp b/core/src/ldml/ldml_markers.cpp
index b1d4604dbc..5d4fb4ff46 100644
--- a/core/src/ldml/ldml_markers.cpp
+++ b/core/src/ldml/ldml_markers.cpp
@@ -268,15 +268,14 @@ add_pending_markers(
marker_map *markers,
marker_list &last_markers,
const std::u32string::const_iterator &last,
- const std::u32string::const_iterator &end,
- const icu::Normalizer2 *nfd) {
+ const std::u32string::const_iterator &end) {
// quick check to see if there's no work to do.
if(markers == nullptr) {
return;
}
/** which character this marker is 'glued' to. */
char32_t marker_ch;
- icu::UnicodeString decomposition;
+ std::u32string decomposition;
if (last == end) {
// at end of text, so use a special value to indicate 'EOT'.
marker_ch = MARKER_BEFORE_EOT;
@@ -285,17 +284,13 @@ add_pending_markers(
// if the character is composed, we need to use the first decomposed char
// as the 'glue'.
- if(!nfd->getDecomposition(ch, decomposition)) {
+ if(!km::core::util::normalize_nfd(ch, decomposition)) {
// char does not have a decomposition - so it may be used for the glue
- marker_ch = ch;
- decomposition.remove(); // no other entries needed
- } else {
- // 'glue' is the first codepoint of the decomposition.
- marker_ch = decomposition.char32At(0);
- if (decomposition.countChar32() == 1) {
- decomposition.remove(); // no other entries needed
- } // else: will add the remainder below
+ // the 'if' is only for the assertions here.
+ assert(decomposition.length() == 1); // should be a single UTF-32 char
+ assert(decomposition.at(0) == ch); // should be the same char
}
+ marker_ch = decomposition.at(0); // always the first char
}
markers->emplace_back(marker_ch);
// now, update the map with these markers (in order) on this character.
@@ -304,10 +299,10 @@ add_pending_markers(
markers->emplace_back(marker_ch, *i);
}
// add any further entries due to decomposition
- if (!decomposition.isEmpty()) {
- // We already added the base char above, add teh rest
- for (auto i=1; iemplace_back(decomposition.char32At(i));
+ if (decomposition.length() > 1) {
+ // We already added the base char above, add the rest
+ for (size_t i=1; iemplace_back(decomposition.at(i));
}
}
// clear the list
@@ -318,10 +313,6 @@ std::u32string
remove_markers(const std::u32string &str, marker_map *markers, marker_encoding encoding) {
std::u32string out;
marker_list last_markers;
- UErrorCode status = U_ZERO_ERROR;
- const icu::Normalizer2 *nfd = icu::Normalizer2::getNFDInstance(status);
- UASSERT_SUCCESS(status);
-
auto last = str.begin(); // points to the part of the string after the last matched marker
for (auto i = str.begin(); i != str.end();) {
auto marker_no = parse_next_marker(i, str.end(), encoding);
@@ -329,7 +320,7 @@ remove_markers(const std::u32string &str, marker_map *markers, marker_encoding e
// add any markers found before this entry, but only if there is intervening
// text. This prevents the sentinel or the '\u' from becoming the attachment char.
if (i != last) {
- add_pending_markers(markers, last_markers, last, str.end(), nfd);
+ add_pending_markers(markers, last_markers, last, str.end());
out.append(last, i); // append any non-marker text since the end of the last marker
last = i; // advance over text we've already appended
}
@@ -347,7 +338,7 @@ remove_markers(const std::u32string &str, marker_map *markers, marker_encoding e
// add any remaining pending markers.
// if last == str.end() then this wil be MARKER_BEFORE_EOT
// otherwise it will be the glue character
- add_pending_markers(markers, last_markers, last, str.end(), nfd);
+ add_pending_markers(markers, last_markers, last, str.end());
// get the suffix between the last marker and the end (could be nothing)
out.append(last, str.end());
return out;
diff --git a/core/src/ldml/ldml_markers.hpp b/core/src/ldml/ldml_markers.hpp
index cfd98d1b15..bbec2f23cc 100644
--- a/core/src/ldml/ldml_markers.hpp
+++ b/core/src/ldml/ldml_markers.hpp
@@ -17,10 +17,6 @@
#include "debuglog.h"
#include "core_icu.h"
-#include "unicode/uniset.h"
-#include "unicode/usetiter.h"
-#include "unicode/regex.h"
-#include "unicode/utext.h"
namespace km {
namespace core {
diff --git a/core/src/ldml/ldml_transforms.cpp b/core/src/ldml/ldml_transforms.cpp
index 7b161ff4f9..8e37011bda 100644
--- a/core/src/ldml/ldml_transforms.cpp
+++ b/core/src/ldml/ldml_transforms.cpp
@@ -437,20 +437,14 @@ reorder_group::apply(std::u32string &str) const {
}
transform_entry::transform_entry(const transform_entry &other)
- : fFrom(other.fFrom), fTo(other.fTo), fFromPattern(nullptr), fMapFromStrId(other.fMapFromStrId),
+ : fFrom(other.fFrom), fTo(other.fTo), fFromPattern(other.fFromPattern), fMapFromStrId(other.fMapFromStrId),
fMapToStrId(other.fMapToStrId), fMapFromList(other.fMapFromList), fMapToList(other.fMapToList),
normalization_disabled(other.normalization_disabled) {
- if (other.fFromPattern) {
- // clone pattern
- fFromPattern.reset(other.fFromPattern->clone());
- }
}
transform_entry::transform_entry(const std::u32string &from, const std::u32string &to)
- : fFrom(from), fTo(to), fFromPattern(nullptr), fMapFromStrId(), fMapToStrId(), fMapFromList(), fMapToList(), normalization_disabled(false) {
+ : fFrom(from), fTo(to), fFromPattern(from), fMapFromStrId(), fMapToStrId(), fMapFromList(), fMapToList(), normalization_disabled(false) {
assert(!fFrom.empty());
-
- init();
}
transform_entry::transform_entry(
@@ -461,7 +455,7 @@ transform_entry::transform_entry(
const kmx::kmx_plus &kplus,
bool &valid,
bool norm_disabled)
- : fFrom(from), fTo(to), fFromPattern(nullptr), fMapFromStrId(mapFrom), fMapToStrId(mapTo), normalization_disabled(norm_disabled) {
+ : fFrom(from), fTo(to), fFromPattern(), fMapFromStrId(mapFrom), fMapToStrId(mapTo), normalization_disabled(norm_disabled) {
if (!valid)
return; // exit early
assert(!fFrom.empty()); // TODO-LDML: should not happen?
@@ -469,7 +463,12 @@ transform_entry::transform_entry(
assert(kplus.strs != nullptr);
assert(kplus.vars != nullptr);
assert(kplus.elem != nullptr);
- if(!init()) {
+ std::u32string from2 = fFrom;
+ if (!normalization_disabled) {
+ // normalize, including markers, for regex
+ normalize_nfd_markers(from2, regex_sentinel);
+ }
+ if (!fFromPattern.init(from2)) {
valid = false;
}
@@ -506,160 +505,17 @@ transform_entry::transform_entry(
}
}
-bool
-transform_entry::init() {
- if (fFrom.empty()) {
- return false;
- }
- // TODO-LDML: if we have mapFrom, may need to do other processing.
- std::u32string from2 = fFrom;
- if (!normalization_disabled) {
- // normalize, including markers, for regex
- normalize_nfd_markers(from2, regex_sentinel);
- }
- std::u16string patstr = km::core::kmx::u32string_to_u16string(from2);
- UErrorCode status = U_ZERO_ERROR;
- /* const */ icu::UnicodeString patustr = icu::UnicodeString(patstr.data(), (int32_t)patstr.length());
- // add '$' to match to end
- patustr.append(u'$'); // TODO-LDML: may need to escape some markers. Marker #91 will look like a `[` to the pattern
- fFromPattern.reset(icu::RegexPattern::compile(patustr, 0, status));
- return (UASSERT_SUCCESS(status));
-}
-
size_t
transform_entry::apply(const std::u32string &input, std::u32string &output) const {
- assert(fFromPattern);
- // TODO-LDML: Really? can't go from u32 to UnicodeString?
- // TODO-LDML: Also, we could cache the u16 string at the transformGroup level or higher.
- UErrorCode status = U_ZERO_ERROR;
- const std::u16string matchstr = km::core::kmx::u32string_to_u16string(input);
- icu::UnicodeString matchustr = icu::UnicodeString(matchstr.data(), (int32_t)matchstr.length());
- // TODO-LDML: create a new Matcher every time. These could be cached and reset.
- std::unique_ptr matcher(fFromPattern->matcher(matchustr, status));
- if (!UASSERT_SUCCESS(status)) {
- return 0; // TODO-LDML: return error
- }
-
- if (!matcher->find(status)) { // i.e. matches somewhere, in this case at end of str
- return 0; // no match
- }
-
- // TODO-LDML: this is UTF-16 len, not UTF-32 len!!
- // TODO-LDML: if we had an underlying UText this would be simpler.
- int32_t matchStart = matcher->start(status);
- int32_t matchEnd = matcher->end(status);
- if (!UASSERT_SUCCESS(status)) {
- return 0; // TODO-LDML: return error
- }
- // extract..
- const icu::UnicodeString substr = matchustr.tempSubStringBetween(matchStart, matchEnd);
- // preflight to UTF-32 to get length
- UErrorCode substrStatus = U_ZERO_ERROR; // throwaway status
- // we need the UTF-32 matchLen for our return.
- auto matchLen = substr.toUTF32(nullptr, 0, substrStatus);
-
- // should have matched something.
- assert(matchLen > 0);
-
- // now, do the replace.
-
- /** this is the 'to' or other replacement string.*/
- icu::UnicodeString rustr;
- if (fMapFromStrId == 0) {
- // Normal case: not a map.
- // This replace will apply $1, $2 etc.
- // Convert the fTo into u16 TODO-LDML (we could cache this?)
- const std::u16string rstr = km::core::kmx::u32string_to_u16string(fTo);
- rustr = icu::UnicodeString(rstr.data(), (int32_t)rstr.length());
- } else {
- // Set map case: mapping from/to
-
- // we actually need the group(1) string here.
- // this is only the content in parenthesis ()
- icu::UnicodeString group1 = matcher->group(1, status);
- if (!UASSERT_SUCCESS(status)) {
- // TODO-LDML: could be a malformed from pattern
- return 0; // TODO-LDML: return error
- }
- // now, how long is group1 in UTF-32, hmm?
- UErrorCode preflightStatus = U_ZERO_ERROR; // throwaway status
- auto group1Len = group1.toUTF32(nullptr, 0, preflightStatus);
- char32_t *s = new char32_t[group1Len + 1];
- assert(s != nullptr); // TODO-LDML: OOM
- // convert
- group1.toUTF32((UChar32 *)s, group1Len + 1, status);
- if (!UASSERT_SUCCESS(status)) {
- return 0; // TODO-LDML: memory issue
- }
- std::u32string match32(s, group1Len); // taken from just group1
- // clean up buffer
- delete [] s;
-
- // Now we're ready to do the actual mapping.
-
- // 1., we need to find the index in the source set.
- auto matchIndex = findIndexFrom(match32);
- assert(matchIndex != -1L); // TODO-LDML: not matching shouldn't happen, the regex wouldn't have matched.
- // we already asserted on load that the from and to sets have the same cardinality.
-
- // 2. get the target string, convert to utf-16
- // we use the same matchIndex that was just found
- const std::u16string rstr = km::core::kmx::u32string_to_u16string(fMapToList.at(matchIndex));
-
- // 3. update the UnicodeString for replacement
- rustr = icu::UnicodeString(rstr.data(), (int32_t)rstr.length());
- // and we return to the regular code flow.
- }
- // here we replace the match output. No normalization, yet.
- icu::UnicodeString entireOutput = matcher->replaceFirst(rustr, status);
- if (!UASSERT_SUCCESS(status)) {
- // TODO-LDML: could fail here due to bad input (syntax err)
- return 0;
- }
- // entireOutput includes all of 'input', but modified. Need to substring it.
- icu::UnicodeString outu = entireOutput.tempSubString(matchStart);
-
- // Special case if there's no output, save some allocs
- if (outu.length() == 0) {
- output.clear();
- } else {
- // TODO-LDML: All we are trying to do is to extract the output string. Probably too many steps.
- UErrorCode preflightStatus = U_ZERO_ERROR;
- // calculate how big the buffer is
- auto out32len = outu.toUTF32(nullptr, 0, preflightStatus); // preflightStatus will be an err, because we know the buffer overruns zero bytes
- // allocate
- std::unique_ptr s(new char32_t[out32len + 1]);
- assert(s);
- if (!s) {
- return 0; // TODO-LDML: allocation failed
- }
- // convert
- outu.toUTF32((UChar32 *)(s.get()), out32len + 1, status);
- if (!UASSERT_SUCCESS(status)) {
- return 0; // TODO-LDML: memory issue
- }
- output.assign(s.get(), out32len);
- // NOW do a marker-safe normalize
- if (!normalization_disabled && !normalize_nfd_markers(output)) {
+ auto result = fFromPattern.apply(input, output, fTo, fMapFromList, fMapToList);
+ // NOW do a marker-safe normalize
+ if (result != 0 && !output.empty() && !normalization_disabled) {
+ if (!normalize_nfd_markers(output)) {
DebugLog("normalize_nfd_markers(output) failed");
- return 0; // TODO-LDML: normalization failed.
+ return 0; // TODO-LDML: normalization failed.
}
}
- return matchLen;
-}
-
-int32_t transform_entry::findIndexFrom(const std::u32string &match) const {
- return findIndex(match, fMapFromList);
-}
-
-int32_t transform_entry::findIndex(const std::u32string &match, const std::deque list) {
- int32_t index = 0;
- for(auto e = list.begin(); e < list.end(); e++, index++) {
- if (match == *e) {
- return index;
- }
- }
- return -1; // not found
+ return result;
}
any_group::any_group(const transform_group &g) : type(any_group_type::transform), transform(g), reorder() {
diff --git a/core/src/ldml/ldml_transforms.hpp b/core/src/ldml/ldml_transforms.hpp
index 1b56a76246..90dd5f5e43 100644
--- a/core/src/ldml/ldml_transforms.hpp
+++ b/core/src/ldml/ldml_transforms.hpp
@@ -16,11 +16,7 @@
#include
#include "debuglog.h"
-#include "core_icu.h"
-#include "unicode/uniset.h"
-#include "unicode/usetiter.h"
-#include "unicode/regex.h"
-#include "unicode/utext.h"
+#include "util_regex.hpp"
namespace km {
namespace core {
@@ -111,14 +107,12 @@ public:
private:
const std::u32string fFrom;
const std::u32string fTo;
- std::unique_ptr fFromPattern;
+ km::core::util::km_regex fFromPattern;
const KMX_DWORD fMapFromStrId;
const KMX_DWORD fMapToStrId;
std::deque fMapFromList;
std::deque fMapToList;
- /** Internal function to setup pattern string @returns true on success */
- bool init();
bool normalization_disabled;
/** @returns the index of the item in the fMapFromList list, or -1 */
int32_t findIndexFrom(const std::u32string &match) const;
diff --git a/core/src/meson.build b/core/src/meson.build
index e8008fb78a..212b56f1c2 100644
--- a/core/src/meson.build
+++ b/core/src/meson.build
@@ -16,11 +16,6 @@ if cpp_compiler.get_id() == 'msvc'
version_res += import('windows').compile_resources('version.rc', args:['/n','/c65001'])
endif
-if cpp_compiler.get_id() == 'emscripten'
- # TODO: why do we need this defn here?
- defns += ['-DKM_CORE_LIBRARY']
-endif
-
# ICU4C is used for repertoire tests and core implementation
if target_machine.system() == 'linux'
@@ -36,10 +31,42 @@ else
icu_i18n = icu4c.get_variable('icui18n_dep')
endif
+if cpp_compiler.get_id() == 'emscripten'
+ # TODO: why do we need this defn here?
+ defns += ['-DKM_CORE_LIBRARY']
+ icu_if_not_on_wasm = []
+else
+ # only include this if NOT on wasm.
+ icu_if_not_on_wasm = [icu_uc, icu_i18n]
+endif
+
if icu_uc.found()
defns += '-DHAVE_ICU4C'
endif
+# On wasm, generate util_normalize_table.h automatically from ICU
+generated_headers = []
+
+if cpp_compiler.get_id() == 'emscripten'
+
+util_normalize_table_generator = executable('util_normalize_table_generator',
+ ['util_normalize_table_generator.cpp'],
+ cpp_args: defns + warns,
+ include_directories: [inc],
+ link_args: links,
+ dependencies: [icu_uc, icu_i18n],
+ )
+
+util_normalize_table_h = custom_target('util_normalize_table.h',
+ output: 'util_normalize_table.h',
+ command: [util_normalize_table_generator],
+ capture:true)
+
+generated_headers += util_normalize_table_h
+
+
+endif
+
kmx_files = files(
'actions_normalize.cpp',
@@ -60,6 +87,8 @@ kmx_files = files(
'km_core_processevent_api.cpp',
'jsonpp.cpp',
'util_normalize.cpp',
+ 'util_regex.cpp',
+ 'core_icu.cpp',
'ldml/ldml_processor.cpp',
'ldml/ldml_transforms.cpp',
'ldml/ldml_markers.cpp',
@@ -110,18 +139,19 @@ lib = library('keymancore',
kmx_files,
mock_files,
version_res,
+ generated_headers,
cpp_args: defns + warns + flags,
link_args: links,
version: lib_version,
include_directories: inc,
pic: true,
install: true,
- dependencies: [icu_uc, icu_i18n],
+ dependencies: icu_if_not_on_wasm,
)
headerdirs = [ '.', 'keyman' ] # subdirectories of ${prefix}/include to add to header path
-keymancore = declare_dependency(link_with: lib, include_directories: inc, dependencies: [icu_uc, icu_i18n])
+keymancore = declare_dependency(link_with: lib, include_directories: inc, dependencies: icu_if_not_on_wasm)
pkg = import('pkgconfig')
pkg.generate(
diff --git a/core/src/util_normalize.cpp b/core/src/util_normalize.cpp
index c21b75ea5b..bcdfc1bb3f 100644
--- a/core/src/util_normalize.cpp
+++ b/core/src/util_normalize.cpp
@@ -13,6 +13,7 @@
#ifdef __EMSCRIPTEN__
#include
#include "utfcodec.hpp"
+#include
// JS implementations
EM_JS(char*, NormalizeNFD, (const char* input), {
@@ -21,6 +22,17 @@ EM_JS(char*, NormalizeNFD, (const char* input), {
const nfd = instr.normalize("NFD");
return stringToNewUTF8(nfd);
});
+
+EM_JS(char*, NormalizeNFC, (const char* input), {
+ if (!input) return input; // pass through null
+ const instr = Module.UTF8ToString(input);
+ const nfd = instr.normalize("NFC");
+ return stringToNewUTF8(nfd);
+});
+
+// pull in the generated table
+#include "util_normalize_table.h"
+
#endif
namespace km {
@@ -28,31 +40,15 @@ namespace core {
namespace util {
#ifndef __EMSCRIPTEN__
-
-/**
- * Internal function to normalize with a specified mode.
- * Note: that this function _does_ assert failure, so it is not
- * required to assert its return code. The return is provided so
- * that callers can exit (such as making no change) if there was failure.
- *
- * Also note that "failure" here is something catastrophic: ICU not initialized,
- * or, more likely, some low memory situation. Does not fail on "bad" data.
- * @param n the ICU Normalizer to use
- * @param str input/output string
- * @param status error code, must be initialized on input
- * @return false if failure
- */
-static bool normalize(const icu::Normalizer2 *n, std::u16string &str, UErrorCode &status) {
+inline const icu::Normalizer2 *getNFD(UErrorCode &status) {
+ const icu::Normalizer2 *nfd = icu::Normalizer2::getNFDInstance(status);
UASSERT_SUCCESS(status);
- assert(n != nullptr);
- icu::UnicodeString dest;
- icu::UnicodeString src = icu::UnicodeString(str.data(), (int32_t)str.length());
- n->normalize(src, dest, status);
- // the next line here will assert
- if (UASSERT_SUCCESS(status)) {
- str.assign(dest.getBuffer(), dest.length());
- }
- return U_SUCCESS(status);
+ return nfd;
+}
+inline const icu::Normalizer2 *getNFC(UErrorCode &status) {
+ const icu::Normalizer2 *nfc = icu::Normalizer2::getNFCInstance(status);
+ UASSERT_SUCCESS(status);
+ return nfc;
}
#endif
@@ -66,6 +62,16 @@ bool normalize_nfd(std::u32string &str) {
}
}
+bool normalize_nfc(std::u32string &str) {
+ std::u16string rstr = km::core::kmx::u32string_to_u16string(str);
+ if(!km::core::util::normalize_nfc(rstr)) {
+ return false;
+ } else {
+ str = km::core::kmx::u16string_to_u32string(rstr);
+ return true;
+ }
+}
+
bool normalize_nfd(std::u16string &str) {
#ifdef __EMSCRIPTEN__
std::string instr = convert(str);
@@ -81,9 +87,26 @@ bool normalize_nfd(std::u16string &str) {
return true;
#else
UErrorCode status = U_ZERO_ERROR;
- const icu::Normalizer2 *nfd = icu::Normalizer2::getNFDInstance(status);
- UASSERT_SUCCESS(status);
- return normalize(nfd, str, status);
+ return normalize(getNFD(status), str, status);
+#endif
+}
+
+bool normalize_nfc(std::u16string &str) {
+#ifdef __EMSCRIPTEN__
+ std::string instr = convert(str);
+ const char *in = instr.c_str();
+ char *out = NormalizeNFC(in);
+ if (out == nullptr) {
+ assert(out != nullptr);
+ return false;
+ }
+ std::string outstr(out);
+ str = convert(outstr);
+ free(out);
+ return true;
+#else
+ UErrorCode status = U_ZERO_ERROR;
+ return normalize(getNFC(status), str, status);
#endif
}
@@ -91,26 +114,130 @@ bool normalize_nfd(std::u16string &str) {
* Normalize the input string using ICU, out of place
*/
bool normalize_nfd(km_core_cu const * src, std::u16string &dst) {
- UErrorCode icu_status = U_ZERO_ERROR;
- const icu::Normalizer2 *nfd = icu::Normalizer2::getNFDInstance(icu_status);
- assert(U_SUCCESS(icu_status));
- if(!U_SUCCESS(icu_status)) {
- // TODO: log the failure code
+#ifdef __EMSCRIPTEN__
+ dst = std::u16string(src);
+ return normalize_nfd(dst); // vector to above fcn
+#else
+ UErrorCode status = U_ZERO_ERROR;
+ auto nfd = getNFD(status);
+ if (nfd == nullptr) {
return false;
}
icu::UnicodeString udst;
icu::UnicodeString usrc = icu::UnicodeString(src);
- nfd->normalize(usrc, udst, icu_status);
- assert(U_SUCCESS(icu_status));
- if(!U_SUCCESS(icu_status)) {
- // TODO: log the failure code
+ nfd->normalize(usrc, udst, status);
+ if(!UASSERT_SUCCESS(status)) {
return false;
}
dst.assign(udst.getBuffer(), udst.length());
return true;
+#endif
}
+bool
+normalize_nfd(km_core_usv cp, std::u32string &dst) {
+ // set the output string to the original string
+ dst.clear();
+ dst.append(1, cp);
+#ifdef __EMSCRIPTEN__
+ auto str16 = convert(dst);
+ if (!normalize_nfd(str16)) {
+ return false; // failed, retain original str
+ } else {
+ dst = convert(str16);
+ return true;
+ }
+#else
+ UErrorCode icu_status = U_ZERO_ERROR;
+ const icu::Normalizer2 *nfd = icu::Normalizer2::getNFDInstance(icu_status);
+ assert(U_SUCCESS(icu_status));
+ if (!U_SUCCESS(icu_status)) {
+ // TODO: log the failure code
+ return false;
+ }
+ icu::UnicodeString decomposition;
+ if (!nfd->getDecomposition(cp, decomposition)) {
+ return false; // no error, just no decomposition
+ } else {
+ dst.clear();
+ auto len = decomposition.countChar32();
+ for (int i = 0; i < len; i++) {
+ dst.append(1, decomposition.char32At(i));
+ }
+ return true;
+ }
+#endif
+}
+
+bool is_nfd(const std::u16string& str) {
+#ifdef __EMSCRIPTEN__
+ std::u16string o = str;
+ normalize_nfd(o);
+ return (o == str); // false if changed
+#else
+ UErrorCode status = U_ZERO_ERROR;
+ auto nfd = getNFD(status);
+ if (nfd == nullptr) return false;
+ auto ustr = icu::UnicodeString(false, str.c_str(), (int)str.length());
+ auto result = nfd->isNormalized(ustr, status);
+ if (!UASSERT_SUCCESS(status)) {
+ return false;
+ } else {
+ return result;
+ }
+#endif
+}
+
+bool is_nfd(const std::u32string& str) {
+#ifdef __EMSCRIPTEN__
+ std::u32string o = str;
+ normalize_nfd(o);
+ return (o == str); // false if changed
+#else
+ UErrorCode status = U_ZERO_ERROR;
+ auto nfd = getNFD(status);
+ if (nfd == nullptr) return false;
+ auto ustr = icu::UnicodeString::fromUTF32(reinterpret_cast(str.c_str()), (int)str.length());
+ auto result = nfd->isNormalized(ustr, status);
+ if (!UASSERT_SUCCESS(status)) {
+ return false;
+ } else {
+ return result;
+ }
+#endif
+}
+
+bool has_nfd_boundary_before(km_core_usv cp) {
+#ifdef __EMSCRIPTEN__
+// it's a negative table. entries in the table mean returning false. non-entries return true.
+ for (auto i=0;i<(km_noBoundaryBefore_entries*2);i+=2) {
+ auto start = km_noBoundaryBefore[i+0];
+ if (start > cp) return true;
+ auto count = km_noBoundaryBefore[i+1];
+ auto limit = start+count;
+ if (cp >= start && cp < limit) return false;
+ }
+ return true; // fallthrough
+#else
+ UErrorCode status = U_ZERO_ERROR;
+ auto nfd = getNFD(status);
+ if (nfd == nullptr) return false;
+ return nfd->hasBoundaryBefore(cp);
+#endif
+}
+
+/**
+ * Helper to convert std::u32string to a UTF-32 km_core_usv buffer,
+ * nul-terminated.
+ * Parallel to unicode_string_to_usv()
+ * @returns new buffer, caller owns storage
+ */
+km_core_usv *string_to_usv(const std::u32string& src) {
+ return km::core::kmx::u32dup(src.c_str());
+}
+
+
}
}
}
diff --git a/core/src/util_normalize.hpp b/core/src/util_normalize.hpp
index 90ed650b08..6321ffacda 100644
--- a/core/src/util_normalize.hpp
+++ b/core/src/util_normalize.hpp
@@ -14,6 +14,12 @@ namespace km {
namespace core {
namespace util {
+/** Normalize a u32string inplace to NFC. @return false on failure */
+bool normalize_nfc(std::u32string &str);
+
+/** Normalize a u16string inplace to NFC. @return false on failure */
+bool normalize_nfc(std::u16string &str);
+
/** Normalize a u32string inplace to NFD. @return false on failure */
bool normalize_nfd(std::u32string &str);
@@ -23,6 +29,21 @@ bool normalize_nfd(std::u16string &str);
/** normalize src to dst in NFD. @return false on failure */
bool normalize_nfd(km_core_cu const * src, std::u16string &dst);
+/** normalize (decompose) a single cp to string. @return false on failure */
+bool normalize_nfd(km_core_usv cp, std::u32string &dst);
+
+/** @return true if string is already NFD */
+bool is_nfd(const std::u16string& str);
+
+/** @return true if string is already NFD */
+bool is_nfd(const std::u32string& str);
+
+/** @return true if cp can interacts with prior chars */
+bool has_nfd_boundary_before(km_core_usv cp);
+
+/** convenience function, caller owns storage */
+km_core_usv *string_to_usv(const std::u32string& src);
+
}
}
}
diff --git a/core/src/util_normalize_table_generator.cpp b/core/src/util_normalize_table_generator.cpp
new file mode 100644
index 0000000000..994f16d8b3
--- /dev/null
+++ b/core/src/util_normalize_table_generator.cpp
@@ -0,0 +1,108 @@
+/*
+ Copyright: Β© SIL International.
+ Description: Generator for util_normalize_table.h
+ Create Date: 5 Jun 2024
+ Authors: Steven R. Loomis
+
+ util_normalize_table.h is used under wasm by utilities in util_normalize.cpp to implement
+ normalization functions without needing ICU4C linked.
+
+ This generator is invoked automatically by meson as part of the build.
+*/
+
+#include "kmx/kmx_plus.h"
+#include "kmx/kmx_xstring.h"
+
+#define KMN_NO_ICU 0 // we will need ICU..
+
+#include "core_icu.h"
+
+#include
+#include
+#include
+#include
+
+#include
+
+#include
+
+
+
+int
+write_nfd_table() {
+#ifndef __EMSCRIPTEN__
+ std::cerr << "Note: This is unusual - this generator is usually only run under emscripten!" << std::endl;
+#endif
+
+ // We write to stdout instead of to a file to avoid dealing with the filesystem under emscripten.
+
+ std::cerr << "Writing to stdout." << std::endl;
+
+ // write preamble
+ std::cout << "// GENERATED FILE: DO NOT EDIT" << std::endl;
+ std::cout << "//" << std::endl;
+ std::cout << "// util_normalize_table.h is generated by util_normalize_table_generator.cpp" << std::endl;
+ std::cout << "// and used by util_normalize.cpp" << std::endl;
+ std::cout << std::endl;
+ std::cout << "#pragma once" << std::endl;
+ std::cout << "#define KM_HASBOUNDARYBEFORE_UNICODE_VERSION \"" << U_UNICODE_VERSION << "\"" << std::endl;
+ std::cout << "#define KM_HASBOUNDARYBEFORE_ICU_VERSION \"" << U_ICU_VERSION << "\"" << std::endl;
+ std::cout << std::endl;
+ // we're going to need an NFD normalizer
+ UErrorCode status = U_ZERO_ERROR;
+ const icu::Normalizer2 *nfd = icu::Normalizer2::getNFDInstance(status);
+ assert(U_SUCCESS(status));
+
+ // collect the raw list of chars that do NOT have a boundary before them.
+ std::vector noBoundary;
+ for (km_core_usv ch = 0; ch < km::core::kmx::Uni_MAX_CODEPOINT; ch++) {
+ bool bb = nfd->hasBoundaryBefore(ch);
+ assert(!(ch == 0 && !bb)); // assert that we can use U+0000 as a terminator
+ if (bb) continue; //only emit nonboundary
+ noBoundary.push_back(ch);
+ }
+
+ // now, compress these into runs
+ std::vector> runs; // start,len
+
+ km_core_usv first = 0;
+ km_core_usv last = 0;
+ for(auto i = noBoundary.begin(); i <= noBoundary.end(); i++) {
+ if (first == 0) {
+ first = last = *i;
+ } else {
+ last++;
+ if(i == noBoundary.end() || *i != last) {
+ // end of a run
+ runs.emplace_back(first, last - first);
+ if (i != noBoundary.end()) {
+ // setup for next
+ first = last = *i;
+ }
+ }
+ }
+ }
+
+ // finally, write out metadata and the runs themselves.
+ std::cout << "#define km_noBoundaryBefore_entries " << runs.size() << "\n";
+
+ std::cout << "static char32_t km_noBoundaryBefore[km_noBoundaryBefore_entries * 2 ] = {" << std::endl;
+
+ std::cout << "/* start codepoint, count (inclusive), ...range end */" << std::endl;
+
+ for (auto i = runs.begin(); i < runs.end(); i++) {
+ std::cout << "\t0x" << std::hex << i->first << std::dec << ",\t " << i->second << ", // ...0x" << std::hex << (i->first+i->second-1) << std::endl;
+ }
+
+ // termination
+ std::cout << "};" << std::endl;
+ std::cout << "// end" << std::endl;
+ std::cerr << "Wrote " << runs.size() << " runs representing " << noBoundary.size() << " entries." << std::endl;
+ return 0;
+}
+
+int
+main(int /*argc*/, const char * /*argv*/[]) {
+ write_nfd_table();
+ return 0;
+}
diff --git a/core/src/util_regex.cpp b/core/src/util_regex.cpp
new file mode 100644
index 0000000000..f8982fdf65
--- /dev/null
+++ b/core/src/util_regex.cpp
@@ -0,0 +1,352 @@
+/*
+ Copyright: Β© SIL International.
+ Description: Core Regex Utilities - abstract out ICU dependencies
+ Create Date: 5 Jun 2024
+ Authors: Steven R. Loomis
+*/
+
+#include "util_regex.hpp"
+
+#include "core_icu.h"
+#include "kmx/kmx_xstring.h"
+
+#ifdef __EMSCRIPTEN__
+#include
+#include "utfcodec.hpp"
+#include
+// JS implementations
+
+
+/**
+ * RegexMatchLen(pattern, input)
+ * @param pattern the string, sans trailing $, for the pattern
+ * @param input the text to match against
+ * @return the length, in code units, of the matched portion
+ */
+EM_JS(int, RegexMatchLen, (const char* pattern, const char *input), {
+ const DEBUG_JS = false;
+ if (!pattern) return -1;
+ if (!input) return -1;
+ const patternstr = Module.UTF8ToString(pattern) + '$';
+ const inputstr = Module.UTF8ToString(input);
+ const re = new RegExp(patternstr);
+ const result = re.exec(inputstr);
+ if (DEBUG_JS) console.dir({patternstr,inputstr,re,result});
+ if (!result) return 0; // no match
+ const index = result.index;
+ // code unit indices
+ const startIndex = index;
+ const endIndex = inputstr.length;
+ const matchedText = inputstr.substring(startIndex);
+ const matchedCodepoints = [...matchedText];
+ if (DEBUG_JS) console.dir({index, startIndex,endIndex,matchedText,matchedCodepoints});
+ return matchedCodepoints.length;
+});
+
+/**
+ * RegexGroup1(pattern, input)
+ * @param pattern the string, sans trailing $, for the pattern
+ * @param input the text to match against
+ * @return string of group 1 match or null
+ */
+EM_JS(char*, RegexGroup1, (const char* pattern, const char *input), {
+ const DEBUG_JS = false;
+ if (!pattern) return 0;
+ if (!input) return 0;
+ const patternstr = Module.UTF8ToString(pattern) + '$';
+ const inputstr = Module.UTF8ToString(input);
+ const re = new RegExp(patternstr);
+ const result = re.exec(inputstr);
+ if (!result) return 0; // no match
+ const g1 = result[1];
+ if (DEBUG_JS) console.dir({patternstr,inputstr,re,result,g1});
+ if (!g1) return 0;
+ return stringToNewUTF8(g1);
+});
+
+
+/**
+ * RegexSubstitute
+ * @param pattern the string, sans trailing $, for the pattern
+ * @param input the text to match against
+ * @param to the replacement text
+ * @return the entire updated output string
+ */
+EM_JS(char*, RegexSubstitute, (const char* pattern, const char *input, const char *to), {
+ const DEBUG_JS = false;
+ if (!pattern) return -1;
+ if (!input) return -1;
+ const patternstr = Module.UTF8ToString(pattern) + '$';
+ const inputstr = Module.UTF8ToString(input);
+ const tostr = Module.UTF8ToString(to);
+ const re = new RegExp(patternstr);
+ const output = inputstr.replace(re, tostr);
+ return stringToNewUTF8(output);
+});
+
+// pull in the generated table
+#include "util_normalize_table.h"
+
+#endif
+
+
+namespace km {
+namespace core {
+namespace util {
+
+/** find the */
+int32_t km_regex::findIndex(const std::u32string &match, const std::deque &list) {
+ int32_t index = 0;
+ for(auto e = list.begin(); e < list.end(); e++, index++) {
+ if (match == *e) {
+ return index;
+ }
+ }
+ return -1; // not found
+}
+
+km_regex::km_regex()
+#if KMN_NO_ICU
+#else
+ : fPattern(nullptr)
+#endif
+{
+
+}
+
+
+km_regex::km_regex(const km_regex& other)
+#if KMN_NO_ICU
+ : fPattern(other.fPattern)
+#else
+ : fPattern(nullptr)
+#endif
+{
+#if KMN_NO_ICU
+
+#else
+ if (other.fPattern) {
+ // clone pattern
+ fPattern.reset(other.fPattern->clone());
+ }
+#endif
+}
+
+km_regex::km_regex(const std::u32string &pattern)
+#if KMN_NO_ICU
+ : fPattern(pattern)
+#else
+ : fPattern(nullptr)
+#endif
+{
+ init(pattern);
+}
+
+km_regex::~km_regex() {
+
+}
+
+bool km_regex::valid() const {
+#if KMN_NO_ICU
+ return (!fPattern.empty());
+#else
+ // valid if fPattern is present.
+ return !!fPattern;
+#endif
+}
+
+bool km_regex::init(const std::u32string &pattern) {
+#if KMN_NO_ICU
+// The current implementation makes a new regex every time, so we always return true.
+ assert(!pattern.empty());
+ fPattern = pattern;
+ return true;
+#else
+ if (pattern.empty()) {
+ return false;
+ }
+ std::u16string patstr = km::core::kmx::u32string_to_u16string(pattern);
+ UErrorCode status = U_ZERO_ERROR;
+ /* const */ icu::UnicodeString patustr = icu::UnicodeString(patstr.data(), (int32_t)patstr.length());
+ // add '$' to match to end
+ patustr.append(u'$');
+ fPattern.reset(icu::RegexPattern::compile(patustr, 0, status));
+ return (UASSERT_SUCCESS(status));
+#endif
+}
+
+size_t km_regex::apply(const std::u32string &input, std::u32string &output,
+ const std::u32string &to,
+ const std::deque &fromList,
+ const std::deque &toList ) const {
+#if KMN_NO_ICU
+ // length in code points of match from end
+ std::string patstr = convert(fPattern);
+ std::string instr = convert(input);
+
+ /** code units */
+ const auto matchLen = RegexMatchLen(patstr.c_str(), instr.c_str());
+ assert(matchLen != -1); // error
+ if (matchLen == 0) {
+ return 0; // Normal case return: no match
+ }
+ std::u32string rustr; // replacement
+ if (fromList.empty()) {
+ // Normal case: not a map.
+ // This replace will apply $1, $2 etc.
+ rustr = to;
+ } else {
+ // we actually need the group(1) string here.
+ // this is only the content in parenthesis ()
+ char *group1 = RegexGroup1(patstr.c_str(), instr.c_str());
+ assert(group1 != nullptr);
+ const std::string group1str(group1);
+ const std::u32string match32 = convert(group1str);
+ free(group1);
+ // Now we're ready to do the actual mapping.
+
+ // 1., we need to find the index in the source set.
+ auto matchIndex = findIndex(match32, fromList);
+ assert(matchIndex != -1L); // This indicates that the regex and the fromList are out of sync.
+ // we already asserted on load that the from and to sets have the same cardinality.
+
+ // 2. get the target string
+ // we use the same matchIndex that was just found
+ // 3. update the UnicodeString for replacement
+ rustr = toList.at(matchIndex);
+ }
+ std::string rstr = convert(rustr);
+ // here we replace the match output.
+ char *out = RegexSubstitute(patstr.c_str(), instr.c_str(), rstr.c_str());
+ assert(out != nullptr);
+ std::string outstr(out);
+ free(out);
+ output = convert(outstr);
+ // output includes all of 'input', but modified. Need to substring it.
+ /** code units */
+ const auto matchStart = input.length() - matchLen;
+ // remove the unmatched prefix.
+ output.erase(0, matchStart);
+
+ return matchLen;
+#else
+ assert(fPattern);
+ // TODO-LDML: This entire section may have too many conversions and copies. Could be optimized.
+ UErrorCode status = U_ZERO_ERROR;
+ const std::u16string matchstr = km::core::kmx::u32string_to_u16string(input);
+ icu::UnicodeString matchustr = icu::UnicodeString(matchstr.data(), (int32_t)matchstr.length());
+ // TODO-LDML: create a new Matcher every time. These could be cached and reset.
+ std::unique_ptr matcher(fPattern->matcher(matchustr, status));
+ if (!UASSERT_SUCCESS(status)) {
+ return 0;
+ }
+
+ if (!matcher->find(status)) { // i.e. matches somewhere, in this case at end of str
+ return 0; // Normal case return: no match
+ }
+
+ // Note: this is UTF-16 len, not UTF-32 len.
+ int32_t matchStart = matcher->start(status);
+ int32_t matchEnd = matcher->end(status);
+ if (!UASSERT_SUCCESS(status)) {
+ return 0; // TODO-LDML: return error
+ }
+ // extract..
+ const icu::UnicodeString substr = matchustr.tempSubStringBetween(matchStart, matchEnd);
+ // preflight to UTF-32 to get length
+ UErrorCode substrStatus = U_ZERO_ERROR; // throwaway status
+ // we need the UTF-32 matchLen for our return.
+ auto matchLen = substr.toUTF32(nullptr, 0, substrStatus);
+
+ // should have matched something.
+ assert(matchLen > 0);
+
+
+ // now, do the replace.
+
+ /** this is the 'to' or other replacement string.*/
+ icu::UnicodeString rustr;
+ if (fromList.empty()) {
+ // Normal case: not a map.
+ // This replace will apply $1, $2 etc.
+ // Convert the fTo into u16 TODO-LDML (we could cache this?)
+ const std::u16string rstr = km::core::kmx::u32string_to_u16string(to);
+ rustr = icu::UnicodeString(rstr.data(), (int32_t)rstr.length());
+ } else {
+ // Set map case: mapping from/to
+
+ // we actually need the group(1) string here.
+ // this is only the content in parenthesis ()
+ icu::UnicodeString group1 = matcher->group(1, status);
+ if (!UASSERT_SUCCESS(status)) {
+ // TODO-LDML: could be a malformed from pattern
+ return 0; // TODO-LDML: return error
+ }
+ // now, how long is group1 in UTF-32, hmm?
+ UErrorCode preflightStatus = U_ZERO_ERROR; // throwaway status
+ auto group1Len = group1.toUTF32(nullptr, 0, preflightStatus);
+ char32_t *s = new char32_t[group1Len + 1];
+ assert(s != nullptr); // TODO-LDML: OOM
+ // convert
+ group1.toUTF32((UChar32 *)s, group1Len + 1, status);
+ if (!UASSERT_SUCCESS(status)) {
+ return 0; // TODO-LDML: memory issue
+ }
+ std::u32string match32(s, group1Len); // taken from just group1
+ // clean up buffer
+ delete [] s;
+
+ // Now we're ready to do the actual mapping.
+
+ // 1., we need to find the index in the source set.
+ auto matchIndex = findIndex(match32, fromList);
+ assert(matchIndex != -1L); // This indicates that the regex and the fromList are out of sync.
+ // we already asserted on load that the from and to sets have the same cardinality.
+
+ // 2. get the target string, convert to utf-16
+ // we use the same matchIndex that was just found
+ const std::u16string rstr = km::core::kmx::u32string_to_u16string(toList.at(matchIndex));
+
+ // 3. update the UnicodeString for replacement
+ rustr = icu::UnicodeString(rstr.data(), (int32_t)rstr.length());
+ // and we return to the regular code flow.
+ }
+ // here we replace the match output.
+ icu::UnicodeString entireOutput = matcher->replaceFirst(rustr, status);
+ if (!UASSERT_SUCCESS(status)) {
+ // TODO-LDML: could fail here due to bad input (syntax err)
+ return 0;
+ }
+ // entireOutput includes all of 'input', but modified. Need to substring it.
+ icu::UnicodeString outu = entireOutput.tempSubString(matchStart);
+
+ // Special case if there's no output, save some allocs
+ if (outu.length() == 0) {
+ output.clear();
+ } else {
+ // TODO-LDML: All we are trying to do is to extract the output string. Probably too many steps.
+ UErrorCode preflightStatus = U_ZERO_ERROR;
+ // calculate how big the buffer is
+ auto out32len = outu.toUTF32(nullptr, 0, preflightStatus); // preflightStatus will be an err, because we know the buffer overruns zero bytes
+ // allocate
+ std::unique_ptr s(new char32_t[out32len + 1]);
+ assert(s);
+ if (!s) {
+ return 0;
+ }
+ // convert
+ outu.toUTF32((UChar32 *)(s.get()), out32len + 1, status);
+ if (!UASSERT_SUCCESS(status)) {
+ return 0;
+ }
+ output.assign(s.get(), out32len);
+ }
+ return matchLen;
+
+#endif
+}
+
+
+}
+}
+}
diff --git a/core/src/util_regex.hpp b/core/src/util_regex.hpp
new file mode 100644
index 0000000000..b10ddff1cf
--- /dev/null
+++ b/core/src/util_regex.hpp
@@ -0,0 +1,48 @@
+/*
+ Copyright: Β© SIL International.
+ Description: Normalization and Regex utilities
+ Create Date: 23 May 2024
+ Authors: Steven R. Loomis
+*/
+
+#pragma once
+
+#include "core_icu.h"
+#include "keyman_core.h"
+#include
+#include
+
+namespace km {
+namespace core {
+namespace util {
+
+class km_regex {
+public:
+ km_regex();
+ km_regex(const km_regex &other);
+ km_regex(const std::u32string &pattern);
+ ~km_regex();
+ bool init(const std::u32string &pattern);
+
+ size_t apply(
+ const std::u32string &input,
+ std::u32string &output,
+ const std::u32string &to,
+ const std::deque &fromList,
+ const std::deque &toList) const;
+
+ bool valid() const;
+private:
+#if KMN_NO_ICU
+ std::u32string fPattern; // TODO: by value?
+#else
+ std::unique_ptr fPattern;
+#endif
+// utility functions
+ public:
+ static int32_t findIndex(const std::u32string &match, const std::deque &list);
+};
+
+} // namespace util
+} // namespace core
+} // namespace km
diff --git a/core/tests/unit/kmnkbd/meson.build b/core/tests/unit/kmnkbd/meson.build
index 0adda0a35e..7285b9bf11 100644
--- a/core/tests/unit/kmnkbd/meson.build
+++ b/core/tests/unit/kmnkbd/meson.build
@@ -34,7 +34,7 @@ test_path = join_paths(meson.current_build_dir(), '..', 'kmx')
tests_flags = []
if cpp_compiler.get_id() == 'emscripten'
- tests_flags += ['-lnodefs.js', '-sEXPORTED_RUNTIME_METHODS=[\'UTF8ToString\']']
+ tests_flags += ['-lnodefs.js', wasm_exported_runtime_methods]
endif
foreach t : tests
diff --git a/core/tests/unit/ldml/ldml_test_source.cpp b/core/tests/unit/ldml/ldml_test_source.cpp
index 439236fd77..f2cca9ee5f 100644
--- a/core/tests/unit/ldml/ldml_test_source.cpp
+++ b/core/tests/unit/ldml/ldml_test_source.cpp
@@ -18,6 +18,9 @@
#include
+// Ensure that ICU gets included even on wasm.
+#define KMN_IN_LDML_TESTS
+
#include // for char to vk mapping tables
#include // for surrogate pair macros
#include
diff --git a/core/tests/unit/ldml/meson.build b/core/tests/unit/ldml/meson.build
index f164e90f12..81f63033fe 100644
--- a/core/tests/unit/ldml/meson.build
+++ b/core/tests/unit/ldml/meson.build
@@ -126,7 +126,7 @@ test('test_context_normalization', test_context_normalization, suite: 'ldml')
# Build and run additional test_unicode test
test_unicode = executable('test_unicode', 'test_unicode.cpp',
- ['test_unicode.cpp', common_test_files],
+ ['test_unicode.cpp', common_test_files, generated_headers],
cpp_args: defns + warns,
include_directories: [inc, libsrc, '../../../../developer/src/ext/json'],
link_args: links + tests_flags,
diff --git a/core/tests/unit/ldml/test_transforms.cpp b/core/tests/unit/ldml/test_transforms.cpp
index 94c53f3845..fa909516bc 100644
--- a/core/tests/unit/ldml/test_transforms.cpp
+++ b/core/tests/unit/ldml/test_transforms.cpp
@@ -1,11 +1,13 @@
#include "../../../src/ldml/ldml_markers.hpp"
#include "../../../src/ldml/ldml_transforms.hpp"
+#include "../../../src/util_regex.hpp"
#include "kmx/kmx_plus.h"
#include "kmx/kmx_xstring.h"
#include "test_color.h"
#include
#include
#include
+#include "../../../src/util_regex.hpp"
// TODO-LDML: normal asserts wern't working, so using some hacks.
// #include "ldml_test_utils.hpp"
@@ -13,14 +15,14 @@
// #include "debuglog.h"
#ifndef zassert_string_equal
-#define zassert_string_equal(actual, expected) \
- { \
- if (actual != expected) { \
- std::wcerr << __FILE__ << ":" << __LINE__ << ": " << console_color::fg(console_color::BRIGHT_RED) \
- << "got: " << km::core::kmx::Debug_UnicodeString(actual, 0) << " expected " \
- << km::core::kmx::Debug_UnicodeString(expected, 1) << console_color::reset() << std::endl; \
- return EXIT_FAILURE; \
- } \
+#define zassert_string_equal(actual, expected) \
+ { \
+ if (actual != expected) { \
+ std::wcerr << __FILE__ << ":" << __LINE__ << ": " << console_color::fg(console_color::BRIGHT_RED) << "got: " << actual \
+ << " " << km::core::kmx::Debug_UnicodeString(actual, 0) << " expected " << expected << " " \
+ << km::core::kmx::Debug_UnicodeString(expected, 1) << console_color::reset() << std::endl; \
+ return EXIT_FAILURE; \
+ } \
}
#endif
@@ -624,16 +626,16 @@ test_map() {
std::cout << __FILE__ << ":" << __LINE__ << " transform_entry::findIndex" << std::endl;
{
std::deque list;
- assert_equal(transform_entry::findIndex(U"Does Not Exist", list), -1);
+ assert_equal(km::core::util::km_regex::findIndex(U"Does Not Exist", list), -1);
list.emplace_back(U"0th");
list.emplace_back(U"First");
list.emplace_back(U"Second");
- assert_equal(transform_entry::findIndex(U"First", list), 1);
- assert_equal(transform_entry::findIndex(U"0th", list), 0);
- assert_equal(transform_entry::findIndex(U"Second", list), 2);
- assert_equal(transform_entry::findIndex(U"Nowhere", list), -1);
+ assert_equal(km::core::util::km_regex::findIndex(U"First", list), 1);
+ assert_equal(km::core::util::km_regex::findIndex(U"0th", list), 0);
+ assert_equal(km::core::util::km_regex::findIndex(U"Second", list), 2);
+ assert_equal(km::core::util::km_regex::findIndex(U"Nowhere", list), -1);
}
return EXIT_SUCCESS;
@@ -1135,6 +1137,102 @@ test_normalize() {
return EXIT_SUCCESS;
}
+/** test for the util_regex.hpp functions */
+int
+test_util_regex() {
+ std::cout << "== " << __FUNCTION__ << std::endl;
+
+ {
+ std::cout << __FILE__ << ":" << __LINE__ << " * util_regex.hpp null tests" << std::endl;
+ km::core::util::km_regex r;
+ assert(!r.valid()); // not valid because of an empty string
+ }
+ {
+ std::cout << __FILE__ << ":" << __LINE__ << " * util_regex.hpp simple tests" << std::endl;
+ km::core::util::km_regex r(U"ion");
+ assert(r.valid());
+ const std::u32string to(U"ivity");
+ const std::deque fromList;
+ const std::deque toList;
+ std::u32string output;
+ auto apply0 = r.apply(U"not present", output, to, fromList, toList);
+ assert_equal(apply0, 0); // not found
+
+ const std::u32string input(U"action");
+ auto apply1 = r.apply(input, output, to, fromList, toList);
+ assert_equal(apply1, 3); // matched last 3 codepoints
+ std::u32string expect(U"ivity");
+ zassert_string_equal(output, expect)
+ }
+ {
+ std::cout << __FILE__ << ":" << __LINE__ << " * util_regex.hpp wide tests" << std::endl;
+ km::core::util::km_regex r(U"eπ»");
+ assert(r.valid());
+ const std::u32string to(U"π");
+ const std::deque fromList;
+ const std::deque toList;
+ std::u32string output;
+ const std::u32string input(U":eπ»");
+ auto apply1 = r.apply(input, output, to, fromList, toList);
+ assert_equal(apply1, 2); // matched last 2 codepoints
+ std::u32string expect(U"π");
+ zassert_string_equal(output, expect)
+ }
+ {
+ std::cout << __FILE__ << ":" << __LINE__ << " * util_regex.hpp simple map tests" << std::endl;
+ km::core::util::km_regex r(U"(A|B|C)");
+ assert(r.valid());
+ const std::u32string to(U"$[1:alpha2]"); // ignored
+ std::deque fromList;
+ fromList.emplace_back(U"A");
+ fromList.emplace_back(U"B");
+ fromList.emplace_back(U"C");
+ std::deque toList;
+ toList.emplace_back(U"N");
+ toList.emplace_back(U"O");
+ toList.emplace_back(U"P");
+ std::u32string output;
+ auto apply0 = r.apply(U"not present", output, to, fromList, toList);
+ assert_equal(apply0, 0); // not found
+
+ const std::u32string input(U"WHOA");
+ auto apply1 = r.apply(input, output, to, fromList, toList);
+ assert_equal(apply1, 1); // matched last 1 codepoint
+ std::u32string expect(U"N");
+ zassert_string_equal(output, expect)
+ }
+ {
+ std::cout << __FILE__ << ":" << __LINE__ << " * util_regex.hpp wide map tests" << std::endl;
+ km::core::util::km_regex r(U"(π·|π»|ππ|x)");
+ assert(r.valid());
+ const std::u32string to(U"$[1:alpha2]"); // ignored
+ std::deque fromList;
+ fromList.emplace_back(U"π·");
+ fromList.emplace_back(U"π»");
+ fromList.emplace_back(U"ππ");
+ fromList.emplace_back(U"x");
+ std::deque toList;
+ toList.emplace_back(U"x");
+ toList.emplace_back(U"π·");
+ toList.emplace_back(U"π»");
+ toList.emplace_back(U"π");
+ std::u32string output;
+ auto apply0 = r.apply(U"not present", output, to, fromList, toList);
+ assert_equal(apply0, 0); // not found
+
+ assert_equal(r.apply(U"WHOππ·", output, to, fromList, toList), 1);
+ zassert_string_equal(output, U"x");
+
+ assert_equal(r.apply(U"WHOπx", output, to, fromList, toList), 1);
+ zassert_string_equal(output, U"π");
+
+ assert_equal(r.apply(U"WHOππ", output, to, fromList, toList), 2); // 2 codepoints
+ zassert_string_equal(output, U"π»");
+ }
+
+ return EXIT_SUCCESS;
+}
+
int
main(int argc, const char *argv[]) {
int rc = EXIT_SUCCESS;
@@ -1152,6 +1250,9 @@ main(int argc, const char *argv[]) {
console_color::enabled = console_color::isaterminal() || arg_color;
+ if (test_util_regex() != EXIT_SUCCESS) {
+ rc = EXIT_FAILURE;
+ }
if (test_transforms() != EXIT_SUCCESS) {
rc = EXIT_FAILURE;
diff --git a/core/tests/unit/ldml/test_unicode.cpp b/core/tests/unit/ldml/test_unicode.cpp
index 1c2da5444d..b36f1a3e38 100644
--- a/core/tests/unit/ldml/test_unicode.cpp
+++ b/core/tests/unit/ldml/test_unicode.cpp
@@ -10,6 +10,9 @@
#include
#include
+// Ensure that ICU gets included even on wasm.
+#define KMN_IN_LDML_TESTS
+
#include "keyman_core.h"
#include "path.hpp"
@@ -22,6 +25,8 @@
#include
#include
#include "json.hpp"
+#include "util_normalize.hpp"
+#include "kmx/kmx_xstring.h"
#include
#include
@@ -40,6 +45,11 @@
} \
}
+#ifdef __EMSCRIPTEN__
+// Pull this in to verify versions
+#include "util_normalize_table.h"
+#endif
+
//-------------------------------------------------------------------------------------
// Unicode version tests
//-------------------------------------------------------------------------------------
@@ -172,6 +182,42 @@ const std::string &block_unicode_ver) {
std::cout << std::endl;
}
+#ifdef __EMSCRIPTEN__
+inline const char *boolstr(bool b) {
+ return b?"T":"f";
+}
+
+void test_has_boundary_before() {
+ std::cout << "= " << __FUNCTION__ << std::endl;
+ std::cout << "I see we are on Emscripten / wasm! Now we will do some additional tests." << std::endl;
+ std::string icu4c_unicode(U_UNICODE_VERSION), header_unicode(KM_HASBOUNDARYBEFORE_UNICODE_VERSION),
+ icu4c_icu(U_ICU_VERSION), header_icu(KM_HASBOUNDARYBEFORE_ICU_VERSION);
+ std::cout << "Unicode: " << U_UNICODE_VERSION << ", and from the table file: " << KM_HASBOUNDARYBEFORE_UNICODE_VERSION << std::endl;
+ std::cout << "It would be very strange for these versions to be out of sync. Some sort of build or tool problem." << std::endl;
+ assert_basic_equal(icu4c_unicode, header_unicode);
+ assert_basic_equal(icu4c_icu, header_icu);
+
+ std::cout << std::endl << "Now, let's make sure has_nfd_boundary_before() matches ICU." << std::endl;
+
+ UErrorCode status = U_ZERO_ERROR;
+ const icu::Normalizer2 *nfd = icu::Normalizer2::getNFDInstance(status);
+ UASSERT_SUCCESS(status);
+
+ // now, test that hasBoundaryBefore is the same
+ for (km_core_usv cp = 0; cp < km::core::kmx::Uni_MAX_CODEPOINT; cp++) {
+ auto km_hbb = km::core::util::has_nfd_boundary_before(cp);
+ auto icu_hbb = nfd->hasBoundaryBefore(cp);
+
+ if (km_hbb != icu_hbb) {
+ std::cerr << "Error: util_normalize_table.h said " << boolstr(km_hbb) << " but ICU said " << boolstr(icu_hbb) << " for "
+ << "has_nfd_boundary_before(0x" << std::hex << cp << std::dec << ")" << std::endl;
+ }
+ assert(km_hbb == icu_hbb);
+ }
+ std::cout << "All OK!" << std::endl;
+}
+#endif
+
int test_all(const char *jsonpath, const char *packagepath, const char *blockspath) {
std::cout << "= " << __FUNCTION__ << std::endl;
@@ -187,6 +233,10 @@ int test_all(const char *jsonpath, const char *packagepath, const char *blockspa
test_unicode_versions(versions, package, block_unicode_ver);
+#ifdef __EMSCRIPTEN__
+ test_has_boundary_before();
+#endif
+
return EXIT_SUCCESS;
}
diff --git a/developer/src/kmcmplib/include/kmcompx.h b/developer/src/kmcmplib/include/kmcompx.h
index a1403ad02f..7e7b270496 100644
--- a/developer/src/kmcmplib/include/kmcompx.h
+++ b/developer/src/kmcmplib/include/kmcompx.h
@@ -29,6 +29,7 @@ typedef KMX_WCHAR* PKMX_WCHAR ;
#ifndef _MSC_VER
#include
+#include
template < typename T, size_t N >
size_t _countof( T ( & /*arr*/ )[ N ] )
diff --git a/developer/src/kmcmplib/src/CompMsg.cpp b/developer/src/kmcmplib/src/CompMsg.cpp
index 86a9b4d475..70ea4d7b81 100644
--- a/developer/src/kmcmplib/src/CompMsg.cpp
+++ b/developer/src/kmcmplib/src/CompMsg.cpp
@@ -1,11 +1,7 @@
#include
+#include
+
@@ -139,6 +151,24 @@
This corresponds to the following source line:
store(&message) 'Here is a message about a keyboard'
+
+
+ The Keyboard Version documents the version of the keyboard.
+ A keyboard version should be updated whenever there are changes to a keyboard. The good principles to follow are:
+
+ - Increment the major version number for a
+ keyboard that has significant new functionality.
+ - Increment the minor version number for changes that impact functionality but not
+ in a significant manner.
+ - Optionally, use a third number for bug fixes.
+
+
+ This corresponds to the following source line:
+ store(&keyboardversion) '1.1.2'
+ Note: there is a difference between &keyboardversion, which documents the keyboard version, and &version,
+ which determines which version of Keyman a keyboard will run with.
+
+
In this field, enter information about the keyboard for your own reference. These comments will only be visible in the
source file, and not to users of your keyboard.
@@ -183,6 +213,41 @@
Tests your KeymanWeb keyboard in an Internet Explorer embedded window
+
+ The Features grid controls which additional file components are included in
+ the keyboard. Each of the features relates to a system store. Here are the file
+ components:
+
+ - Embedded JavaScript
+ - Embedded CSS
+ - Web Help
+ - Include Codes
+ - Desktop On-Screen Keyboard (auto-included if Targets is any)
+ - Touch-Optimised Keyboard (auto-included if Targets is any)
+
+ Icon will be automatically included when a new keyboard project is created.
+
+
+
+
+ This will open a selection dialog allowing you to choose a feature to add to
+ the keyboard project. Adding a feature will add an extra tab to the editor,
+ and add the corresponding store to the keyboard source
+
+
+
+ Depending on what is included in the Feature Grid, you can select a feature from
+ the grid then click on Edit... This will take you to the corresponding tab and
+ let you make changes.
+
+
+
+
+ Removing a feature will not delete the component file,
+ but will just remove the store from the keyboard source.
+
+
+
@@ -197,8 +262,40 @@
The toolbox allows for the changing of the colours, addition of text and shapes. It also allows for the moving of the icon around the canvas and also a preview of the icon is displayed.
+
+
+ Click on a colour from the box to apply the colour onto the Keyboard's
+ icon. You can see the Foreground Color displays the colour you chose. To deselect
+ the colour, click on the X mark on the corner left of the colour box, or choose
+ another colour.
+
+
+
+ This tab allows you to edit the visual representation of your keyboard
+ layout. The content on this tab is stored in the .kvks file associated with
+ your keyboard. The visual representation is used only in desktop and desktop
+ web; however if no touch layout is defined, this layout will be synthesized
+ into a touch layout automatically.
+
+ An On-Screen keyboard is optional but in most keyboards is recommended.
+ The On-Screen keyboard may not always match the actual layout identically,
+ because you may choose to hide some of the details of encoding from the
+ interface presented to the user.
+
+
+ This keyboard layout can also be printed or included in HTML or other documentation.
+ The editor allows you to export the file to HTML, PNG or BMP formats.
+
+
+
+
+ If this option is checked, when the Fill from layout button is clicked,
+ then keys without corresponding rules in the Layout will be filled with the
+ base layout character.
+
+
@@ -331,33 +428,185 @@
The debugger can be used without the debug information by clicking on Test without debugger.
-
+
+
-
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/developer/src/tike/xml/layoutbuilder/builder.xsl b/developer/src/tike/xml/layoutbuilder/builder.xsl
index 3b75cd81a3..d43d90c574 100644
--- a/developer/src/tike/xml/layoutbuilder/builder.xsl
+++ b/developer/src/tike/xml/layoutbuilder/builder.xsl
@@ -271,6 +271,10 @@
+
+