diff --git a/.jules/bolt.md b/.jules/bolt.md index 5d54a29f7..77119f7cb 100644 --- a/.jules/bolt.md +++ b/.jules/bolt.md @@ -1,3 +1,7 @@ ## 2024-05-18 - Language Detector Optimization **Learning:** `LanguageDetector.detect` uses `Regex("[^\\p{L}]+")` inside `detectByHeuristics`, creating a new Regex instance on every call. This method is called via `ProofreadHelper` which is likely triggered on user typing or interaction. Pre-compiling the Regex as a top-level constant avoids the overhead of regex compilation for every detection call. **Action:** Always check for repeated Regex instantiations in frequently called string processing or detection methods and extract them to top-level or companion object properties. + +## 2024-05-18 - StringUtils Whitespace Split Optimization +**Learning:** `StringUtils.splitOnWhitespace` creates a `Regex("\\s+")` inline every time it is called. Since this is an extension function used across settings and text processing, this causes unnecessary allocations. +**Action:** Abstract frequent inline regex usages into `private val` properties at the object or file level to avoid repeated regex compilation. diff --git a/app/src/main/java/helium314/keyboard/latin/common/StringUtils.kt b/app/src/main/java/helium314/keyboard/latin/common/StringUtils.kt index 51c6edfdd..d0360ff21 100644 --- a/app/src/main/java/helium314/keyboard/latin/common/StringUtils.kt +++ b/app/src/main/java/helium314/keyboard/latin/common/StringUtils.kt @@ -291,7 +291,8 @@ fun isEmoji(c: Int): Boolean = StringUtils.mightBeEmoji(c) && isEmoji(StringUtil /** returns whether the text is a single emoji */ fun isEmoji(text: CharSequence): Boolean = mightBeEmoji(text) && text.matches(emoRegex) -fun String.splitOnWhitespace() = split(Regex("\\s+")).filter { it.isNotEmpty() } +private val WHITESPACE_REGEX = Regex("\\s+") +fun String.splitOnWhitespace() = split(WHITESPACE_REGEX).filter { it.isNotEmpty() } // from https://github.com/mathiasbynens/emoji-test-regex-pattern, MIT license // matches single emojis only