diff --git a/app/src/main/java/it/palsoftware/pastiera/core/suggestions/SuggestionEngine.kt b/app/src/main/java/it/palsoftware/pastiera/core/suggestions/SuggestionEngine.kt index 0083ca2a2..fcbdf1aa3 100644 --- a/app/src/main/java/it/palsoftware/pastiera/core/suggestions/SuggestionEngine.kt +++ b/app/src/main/java/it/palsoftware/pastiera/core/suggestions/SuggestionEngine.kt @@ -28,7 +28,6 @@ class SuggestionEngine( private val accentCache: MutableMap = mutableMapOf() private val tag = "SuggestionEngine" private val wordNormalizeCache: MutableMap = mutableMapOf() - private val accentChars = setOf('à', 'è', 'é', 'ì', 'ò', 'ó', 'ù', 'À', 'È', 'É', 'Ì', 'Ò', 'Ó', 'Ù') // Keyboard layout positions - built dynamically based on layout type private var keyboardPositions: Map> = buildKeyboardPositions("qwerty") @@ -526,7 +525,7 @@ class SuggestionEngine( } } - val hasAccent = entry.word.any { it in accentChars } + val hasAccent = stripAccents(entry.word) != entry.word val hasDigit = entry.word.any { it.isDigit() } val hasSymbol = entry.word.any { !it.isLetterOrDigit() && it != '\'' } val isSameBaseLetter = entry.word.equals(currentWord, ignoreCase = true) diff --git a/app/src/main/java/it/palsoftware/pastiera/core/suggestions/UserNGramStore.kt b/app/src/main/java/it/palsoftware/pastiera/core/suggestions/UserNGramStore.kt index 8e8d3df23..6c029ae2a 100644 --- a/app/src/main/java/it/palsoftware/pastiera/core/suggestions/UserNGramStore.kt +++ b/app/src/main/java/it/palsoftware/pastiera/core/suggestions/UserNGramStore.kt @@ -43,6 +43,37 @@ class UserNGramStore(context: Context) : SQLiteOpenHelper( "CREATE INDEX ${TABLE_BIGRAMS}_lookup ON $TABLE_BIGRAMS " + "($COL_LOCALE, $COL_PREFIX, $COL_COUNT DESC, $COL_LAST_USED DESC)" ) + seedDefaultBigrams(db) + } + + /** + * Pre-populates a small set of very common Vietnamese word-pair predictions, so next-word + * suggestions aren't completely empty for a brand-new install. Seeded at a low baseline + * count (1) so genuinely learned usage (which increments on every real use) naturally + * overtakes it over time rather than permanently dominating. + */ + private fun seedDefaultBigrams(db: SQLiteDatabase) { + val nowMs = System.currentTimeMillis() + db.beginTransaction() + try { + for ((prefix, nextWord) in DEFAULT_VI_BIGRAMS) { + db.insertWithOnConflict( + TABLE_BIGRAMS, + null, + ContentValues().apply { + put(COL_LOCALE, "vi") + put(COL_PREFIX, prefix) + put(COL_NEXT_WORD, nextWord) + put(COL_COUNT, 1) + put(COL_LAST_USED, nowMs) + }, + SQLiteDatabase.CONFLICT_IGNORE + ) + } + db.setTransactionSuccessful() + } finally { + db.endTransaction() + } } override fun onUpgrade(db: SQLiteDatabase, oldVersion: Int, newVersion: Int) { @@ -144,5 +175,137 @@ class UserNGramStore(context: Context) : SQLiteOpenHelper( private const val COL_NEXT_WORD = "next_word" private const val COL_COUNT = "count" private const val COL_LAST_USED = "last_used" + + // prefix is the accent-stripped, lowercase normalized form of the previous word + // (matching NextWordPredictor.normalizedKey); next_word is the real display form. + private val DEFAULT_VI_BIGRAMS: List> = listOf( + "khong" to "biết", + "khong" to "có", + "khong" to "phải", + "khong" to "được", + "khong" to "thể", + "khong" to "sao", + "khong" to "muốn", + "khong" to "còn", + "khong" to "ai", + "khong" to "gì", + "khong" to "đâu", + "khong" to "hiểu", + "khong" to "thích", + "khong" to "bao giờ", + "khong" to "dám", + "toi" to "là", + "toi" to "có", + "toi" to "muốn", + "toi" to "nghĩ", + "toi" to "thích", + "toi" to "đi", + "toi" to "làm", + "toi" to "biết", + "toi" to "không", + "toi" to "sẽ", + "toi" to "đã", + "minh" to "là", + "minh" to "có", + "minh" to "muốn", + "minh" to "đi", + "minh" to "nghĩ", + "minh" to "không", + "ban" to "có", + "ban" to "là", + "ban" to "muốn", + "ban" to "đi", + "ban" to "làm", + "ban" to "ơi", + "anh" to "có", + "anh" to "là", + "anh" to "muốn", + "anh" to "đi", + "anh" to "ơi", + "chi" to "có", + "chi" to "là", + "chi" to "ơi", + "em" to "có", + "em" to "là", + "em" to "muốn", + "em" to "ơi", + "rat" to "vui", + "rat" to "tốt", + "rat" to "nhiều", + "rat" to "đẹp", + "rat" to "thích", + "rat" to "mệt", + "rat" to "tiếc", + "rat" to "khó", + "rat" to "quan trọng", + "kha" to "vui", + "kha" to "tốt", + "kha" to "nhiều", + "qua" to "nhiều", + "qua" to "tốt", + "qua" to "vui", + "co" to "thể", + "co" to "lẽ", + "co" to "người", + "co" to "một", + "co" to "nhiều", + "co" to "vẻ", + "co" to "khi", + "co" to "lúc", + "la" to "một", + "la" to "người", + "la" to "gì", + "la" to "ai", + "hom" to "nay", + "hom" to "qua", + "hom" to "sau", + "bay" to "giờ", + "va" to "tôi", + "va" to "anh", + "va" to "em", + "va" to "các", + "nhung" to "tôi", + "nhung" to "anh", + "nhung" to "không", + "nhung" to "mà", + "xin" to "chào", + "xin" to "lỗi", + "xin" to "cảm ơn", + "cam" to "ơn", + "cam" to "thấy", + "dang" to "làm", + "dang" to "đi", + "dang" to "học", + "se" to "có", + "se" to "là", + "se" to "đi", + "se" to "làm", + "se" to "không", + "da" to "có", + "da" to "là", + "da" to "đi", + "da" to "làm", + "da" to "xong", + "duoc" to "không", + "duoc" to "rồi", + "muon" to "đi", + "muon" to "làm", + "muon" to "biết", + "muon" to "nói", + "noi" to "chuyện", + "noi" to "gì", + "noi" to "với", + "lam" to "gì", + "lam" to "sao", + "lam" to "việc", + "di" to "đâu", + "di" to "học", + "di" to "làm", + "di" to "ngủ", + "an" to "cơm", + "an" to "sáng", + "an" to "trưa", + "an" to "tối" + ) } } diff --git a/app/src/main/java/it/palsoftware/pastiera/inputmethod/telex/VietnameseTelexProcessor.kt b/app/src/main/java/it/palsoftware/pastiera/inputmethod/telex/VietnameseTelexProcessor.kt index e81d9112c..d2fea63b0 100644 --- a/app/src/main/java/it/palsoftware/pastiera/inputmethod/telex/VietnameseTelexProcessor.kt +++ b/app/src/main/java/it/palsoftware/pastiera/inputmethod/telex/VietnameseTelexProcessor.kt @@ -30,21 +30,39 @@ internal object VietnameseTelexProcessor { fun isActiveForLayout(layoutName: String?): Boolean = layoutName == VIETNAMESE_TELEX_LAYOUT_ID fun rewrite(textBeforeCursor: String, keyChar: Char): Rewrite? { + if (!keyChar.isLetter()) return null val lowerKey = keyChar.lowercaseChar() - if (lowerKey !in toneKeys && lowerKey !in shapeKeys) return null val syllableStart = findSyllableStart(textBeforeCursor) - if (syllableStart == textBeforeCursor.length) return null - + if (syllableStart == textBeforeCursor.length) { + toneCancelledPrefix = "" + return null + } val syllable = textBeforeCursor.substring(syllableStart) - val rewritten = when { - lowerKey == 'z' -> clearDiacritics(syllable) - lowerKey in toneByKey -> applyToneKey(syllable, keyChar) - else -> applyShapeKey(syllable, keyChar) - } ?: return null - - if (rewritten == syllable) return null - return Rewrite(replaceCount = syllable.length, replacement = rewritten) + if (toneCancelledPrefix.isNotEmpty() && !syllable.startsWith(toneCancelledPrefix)) { + toneCancelledPrefix = "" + } + + // Only attempt a Vietnamese conversion while this syllable's onset (the consonants + // before its first vowel) is still a plausible Vietnamese onset. This is what protects + // words like "wolf"/"zoo" (invalid onset "w"/"z") from ever getting converted in the + // first place -- and because nothing gets converted for them, there's nothing to correct + // or roll back later. Once a syllable already contains a REAL conversion (a diacritic + // actually applied), we never undo it: a keystroke that doesn't fit just becomes a + // literal character appended after it, so an already-correct word is never corrupted + // (e.g. "biết" + n -> "biếtn", not a reversion to raw letters). + if ((lowerKey in toneKeys || lowerKey in shapeKeys) && (lowerKey == 'd' || establishedOnsetIsValid(syllable))) { + val rewritten = when { + lowerKey == 'z' -> clearDiacritics(syllable) + lowerKey in toneByKey -> applyToneKey(syllable, keyChar) + else -> applyShapeKey(syllable, keyChar) + } + if (rewritten != null && rewritten != syllable) { + return Rewrite(replaceCount = syllable.length, replacement = rewritten) + } + } + + return null } private fun findSyllableStart(text: String): Int { @@ -79,30 +97,41 @@ internal object VietnameseTelexProcessor { val parts = Parts.fromChar(chars[idx]) val trailing = chars.subList(idx + 1, chars.size).joinToString("") val replacement = when (key) { + // For self-doubling shape keys (aa->â, ee->ê, oo->ô), only allow reaching back + // through a trailing consonant CODA if that coda is still "open" (i.e. we're not + // crossing a completed syllable boundary). We approximate this by requiring the + // trailing text be empty or made up entirely of vowels (e.g. "dau"+a->"dau"-with- + // circumflex, where trailing is the vowel "u"). A trailing consonant (e.g. "mam"+a) + // means an earlier syllable already closed, so the new keystroke starts a fresh one. 'a' -> when { - trailing.isNotEmpty() && !isValidVietnameseTail(trailing) -> null + trailing.isNotEmpty() && !trailing.all { Parts.fromChar(it).isVietnameseVowel() } -> null parts.isBase('a') && !parts.hasShape() -> parts.withShape(CIRCUMFLEX).toChar().toString() parts.isBase('a') && parts.hasShape(CIRCUMFLEX) -> "${caseOf(parts.base)}$keyChar" else -> null } 'e' -> when { - trailing.isNotEmpty() && !isValidVietnameseTail(trailing) -> null + trailing.isNotEmpty() && !trailing.all { Parts.fromChar(it).isVietnameseVowel() } -> null parts.isBase('e') && !parts.hasShape() -> parts.withShape(CIRCUMFLEX).toChar().toString() parts.isBase('e') && parts.hasShape(CIRCUMFLEX) -> "${caseOf(parts.base)}$keyChar" else -> null } 'o' -> when { - trailing.isNotEmpty() && !isValidVietnameseTail(trailing) -> null + trailing.isNotEmpty() && !trailing.all { Parts.fromChar(it).isVietnameseVowel() } -> null parts.isBase('o') && !parts.hasShape() -> parts.withShape(CIRCUMFLEX).toChar().toString() parts.isBase('o') && parts.hasShape(CIRCUMFLEX) -> "${caseOf(parts.base)}$keyChar" else -> null } - 'd' -> when (chars[idx]) { - 'd' -> "đ" - 'D' -> "Đ" - // For an already transformed đ/Đ, treat a new d/D as a literal append - // instead of toggling back, so users can continue typing the next letter. - 'đ', 'Đ' -> return syllable + keyChar + // 'd' doubling (dd->đ) requires the two d's to be strictly adjacent: no reaching + // back through any other letter at all (unlike a/e/o/w, there's no legitimate + // Vietnamese case where a 'd' modifier applies at a distance). A third 'd' press + // reverts đ fully back to the literal double letter, matching how the other shape + // keys escape (e.g. ô + o -> oo), confirmed against real device behavior. + 'd' -> when { + trailing.isNotEmpty() -> null + chars[idx] == 'd' -> "đ" + chars[idx] == 'D' -> "Đ" + chars[idx] == 'đ' -> "dd" + chars[idx] == 'Đ' -> "DD" else -> null } 'w' -> when { @@ -150,7 +179,15 @@ internal object VietnameseTelexProcessor { return null } + // Once a tone is explicitly cancelled (same tone key pressed twice in a row), this syllable + // is remembered as "tone-cancelled" -- the person deliberately said "not that", so no + // further tone key (fresh or cycling) should touch it again, even though the visible text + // is now plain and indistinguishable from a vowel that never had a tone attempt at all. + // Cleared at the next word boundary. + private var toneCancelledPrefix: String = "" + private fun applyToneKey(syllable: String, keyChar: Char): String? { + if (toneCancelledPrefix.isNotEmpty() && syllable.startsWith(toneCancelledPrefix)) return null if (hasSeparatedVowelGroups(syllable)) return null val key = keyChar.lowercaseChar() val toneMark = toneByKey[key] ?: return null @@ -160,7 +197,9 @@ internal object VietnameseTelexProcessor { return if (parts.tone == toneMark) { val cleared = parts.withTone(null).toChar() - syllable.substring(0, targetIndex) + cleared + syllable.substring(targetIndex + 1) + keyChar + val result = syllable.substring(0, targetIndex) + cleared + syllable.substring(targetIndex + 1) + keyChar + toneCancelledPrefix = result + result } else { val toned = parts.withTone(toneMark).toChar() syllable.substring(0, targetIndex) + toned + syllable.substring(targetIndex + 1) @@ -199,13 +238,17 @@ internal object VietnameseTelexProcessor { val prevPos = lastPos - 1 val prev = vowelParts.getOrNull(prevPos) - // "oa"/"oe" commonly take tone on 'o'. + // "oa"/"oe" take tone on 'o' only in an OPEN syllable (nothing after the vowel cluster, + // e.g. "hoa"->"hòa"). In a CLOSED syllable with a trailing coda (e.g. "toan"->"toán", + // "thoat"->"thoát"), the tone moves onto the second vowel instead. + val hasCoda = vowelIndices[lastPos] < chars.lastIndex if (prev != null && prev.base.lowercaseChar() == 'o' && last.base.lowercaseChar() in setOf('a', 'e')) { - return vowelIndices[prevPos] + return if (hasCoda) vowelIndices[lastPos] else vowelIndices[prevPos] } - // If the final vowel is a semivowel, prefer the previous vowel. - if (prev != null && last.base.lowercaseChar() in setOf('i', 'y', 'u')) { + // If the final vowel is a semivowel-like glide (i/y/u, and o as in "ao"/"eo"), prefer + // the previous vowel. + if (prev != null && last.base.lowercaseChar() in setOf('i', 'y', 'u', 'o')) { // Exception: "uy" usually carries tone on y. if (!(prev.base.lowercaseChar() == 'u' && last.base.lowercaseChar() == 'y')) { return vowelIndices[prevPos] @@ -275,6 +318,42 @@ internal object VietnameseTelexProcessor { return isValidVietnameseCoda(trailing) } + // The (~25) legal Vietnamese syllable-initial consonant clusters. Anything else (pr, dr, + // str, sh, bl, fr...) never legitimately starts a Vietnamese syllable. + private val validOnsets = setOf( + "b", "c", "ch", "d", "đ", "g", "gh", "gi", "h", "k", "kh", "l", "m", "n", "ng", "ngh", + "nh", "p", "ph", "q", "r", "s", "t", "th", "tr", "v", "x" + ) + + private fun isValidOnsetPrefix(text: String): Boolean { + if (text.isEmpty()) return true + val t = text.lowercase() + return validOnsets.any { it == t || it.startsWith(t) } + } + + private fun isCompleteValidOnset(text: String): Boolean { + if (text.isEmpty()) return true // onsetless syllable (e.g. "anh", "em") is valid + return text.lowercase() in validOnsets + } + + private fun extractOnset(syllable: String): String { + val firstVowelIdx = syllable.indices.firstOrNull { Parts.fromChar(syllable[it]).isVietnameseVowel() } + ?: return syllable + return syllable.substring(0, firstVowelIdx) + } + + // Whether the syllable's already-typed onset (the consonants before its first vowel, if + // any) is still a legal Vietnamese onset. If a vowel has already appeared, the onset is + // "closed" and must be a COMPLETE valid onset; otherwise it just needs to still be a + // growable prefix of one. This runs before ANY shape/tone conversion is attempted -- an + // invalid onset (e.g. "w", "z", "str") means this syllable was never really Vietnamese, so + // every subsequent letter just inserts literally instead of being converted. + private fun establishedOnsetIsValid(syllable: String): Boolean { + val onset = extractOnset(syllable) + val hasVowel = syllable.length > onset.length + return if (hasVowel) isCompleteValidOnset(onset) else isValidOnsetPrefix(onset) + } + private data class Parts( val base: Char, val shape: Char? = null, diff --git a/app/src/test/java/it/palsoftware/pastiera/inputmethod/CandidatesBarControllerTest.kt b/app/src/test/java/it/palsoftware/pastiera/inputmethod/CandidatesBarControllerTest.kt index 46cb8ba06..07cc31fa0 100644 --- a/app/src/test/java/it/palsoftware/pastiera/inputmethod/CandidatesBarControllerTest.kt +++ b/app/src/test/java/it/palsoftware/pastiera/inputmethod/CandidatesBarControllerTest.kt @@ -63,10 +63,26 @@ class CandidatesBarControllerTest { View.MeasureSpec.makeMeasureSpec(2400, View.MeasureSpec.EXACTLY) ) decorView.layout(0, 0, 1080, 2400) - + + // Dispatch window visibility to decorView + decorView.dispatchWindowVisibilityChanged(View.VISIBLE) + + // Measure and layout the inputView itself + inputView.measure( + View.MeasureSpec.makeMeasureSpec(1080, View.MeasureSpec.EXACTLY), + View.MeasureSpec.makeMeasureSpec(2400, View.MeasureSpec.UNSPECIFIED) + ) + inputView.layout(0, 0, inputView.measuredWidth, inputView.measuredHeight) + + // Dispatch visibility to inputView + inputView.dispatchWindowVisibilityChanged(View.VISIBLE) + + // Ensure the window is visible at the system level + activity.window.setDecorFitsSystemWindows(true) + assertTrue(controller.isInputViewActuallyRendered()) } - + @Test fun candidatesViewIsNotCollapsedByConfiguredSoftwareKeyboardMode() { SettingsManager.setSoftwareKeyboardMode( diff --git a/app/src/test/java/it/palsoftware/pastiera/inputmethod/telex/VietnameseTelexProcessorTest.kt b/app/src/test/java/it/palsoftware/pastiera/inputmethod/telex/VietnameseTelexProcessorTest.kt index a1ccd11da..c7eb6489b 100644 --- a/app/src/test/java/it/palsoftware/pastiera/inputmethod/telex/VietnameseTelexProcessorTest.kt +++ b/app/src/test/java/it/palsoftware/pastiera/inputmethod/telex/VietnameseTelexProcessorTest.kt @@ -14,7 +14,7 @@ class VietnameseTelexProcessorTest { assertRewrite("mo", 'w', "mơ") assertRewrite("tu", 'w', "tư") assertRewrite("d", 'd', "đ") - assertRewrite("đ", 'd', "đd") + assertRewrite("đ", 'd', "dd") } @Test