From bb9afec0e3bdd4d54bacbb3b587377d3871f3790 Mon Sep 17 00:00:00 2001 From: LeanBitLab <245915690+LeanBitLab@users.noreply.github.com> Date: Wed, 7 Oct 2026 11:43:38 +0000 Subject: [PATCH] Extract Regex instances to pre-compiled properties in dictionary classes Extracted inline `Regex("\\s+")` instances within the processing loops of `AppsBinaryDictionary` and `UserBinaryDictionary` to static `SPACE_REGEX` companion object properties. This eliminates the runtime regex compilation overhead that would otherwise execute on a per-word or per-app basis. --- .jules/bolt.md | 5 ++++- .../keyboard/latin/dictionary/AppsBinaryDictionary.kt | 3 ++- .../keyboard/latin/dictionary/UserBinaryDictionary.kt | 3 ++- 3 files changed, 8 insertions(+), 3 deletions(-) diff --git a/.jules/bolt.md b/.jules/bolt.md index 77119f7cb..c822df0bb 100644 --- a/.jules/bolt.md +++ b/.jules/bolt.md @@ -1,7 +1,10 @@ ## 2024-05-18 - Language Detector Optimization **Learning:** `LanguageDetector.detect` uses `Regex("[^\\p{L}]+")` inside `detectByHeuristics`, creating a new Regex instance on every call. This method is called via `ProofreadHelper` which is likely triggered on user typing or interaction. Pre-compiling the Regex as a top-level constant avoids the overhead of regex compilation for every detection call. **Action:** Always check for repeated Regex instantiations in frequently called string processing or detection methods and extract them to top-level or companion object properties. - ## 2024-05-18 - StringUtils Whitespace Split Optimization **Learning:** `StringUtils.splitOnWhitespace` creates a `Regex("\\s+")` inline every time it is called. Since this is an extension function used across settings and text processing, this causes unnecessary allocations. **Action:** Abstract frequent inline regex usages into `private val` properties at the object or file level to avoid repeated regex compilation. + +## 2024-05-18 - Pre-compile Regex in hot dictionary loops +**Learning:** Instantiating `Regex` objects inside dictionary processing loops (like adding words in `UserBinaryDictionary` or `AppsBinaryDictionary`) causes unnecessary allocations and compilation overhead on every item iteration. +**Action:** Always pre-compile `Regex` objects as `private val` properties in the class or companion object to eliminate per-iteration compilation and allocation pressure. diff --git a/app/src/main/java/helium314/keyboard/latin/dictionary/AppsBinaryDictionary.kt b/app/src/main/java/helium314/keyboard/latin/dictionary/AppsBinaryDictionary.kt index 45ae8dea0..a53ad546e 100644 --- a/app/src/main/java/helium314/keyboard/latin/dictionary/AppsBinaryDictionary.kt +++ b/app/src/main/java/helium314/keyboard/latin/dictionary/AppsBinaryDictionary.kt @@ -35,6 +35,7 @@ class AppsBinaryDictionary private constructor( private const val TAG = "AppsBinaryDictionary" private const val NAME = "apps" + private val SPACE_REGEX = Regex("\\s+") private const val FREQUENCY_FOR_APPS = 100 private const val FREQUENCY_FOR_APPS_BIGRAM = 200 @@ -75,7 +76,7 @@ class AppsBinaryDictionary private constructor( var ngramContext = NgramContext.getEmptyPrevWordsContext( BinaryDictionary.MAX_PREV_WORD_COUNT_FOR_N_GRAM ) - for (word in appLabel.split(Regex("\\s+"))) { + for (word in appLabel.split(SPACE_REGEX)) { if (word.isEmpty()) continue if (DEBUG_DUMP) { Log.d(TAG, "addName word = $word") diff --git a/app/src/main/java/helium314/keyboard/latin/dictionary/UserBinaryDictionary.kt b/app/src/main/java/helium314/keyboard/latin/dictionary/UserBinaryDictionary.kt index 1cebf2439..f65a39bea 100644 --- a/app/src/main/java/helium314/keyboard/latin/dictionary/UserBinaryDictionary.kt +++ b/app/src/main/java/helium314/keyboard/latin/dictionary/UserBinaryDictionary.kt @@ -79,6 +79,7 @@ class UserBinaryDictionary protected constructor( Words.FREQUENCY ) + private val SPACE_REGEX = Regex("\\s+") private const val NAME = "userunigram" fun getDictionary( @@ -195,7 +196,7 @@ class UserBinaryDictionary protected constructor( ) } // ponytail: split phrase into unigrams and n-grams for next-word prediction - val parts = word.split(Regex("\\s+")) + val parts = word.split(SPACE_REGEX) if (parts.size > 1) { for (part in parts) { if (part.length <= MAX_WORD_LENGTH && part.isNotEmpty() && part != word) {