/* * Copyright 2026 Charles KWON (KWON Ohjun) * charleskwon@medithings.co.kr / charleskwonohjun@gmail.com * MEDiThings Inc. * All rights reserved. */ package com.example.medilightv2android.speech /** * Local (offline) parser that extracts a volume in milliliters from speech-recognition text. * * Supported patterns: * - Arabic digits with unit: "150ml", "200 미리" * - Korean numerals with unit: "백오십 밀리리터", "이백 미리" * - English numerals with unit: "one hundred fifty ml", "two hundred milliliters" * - English digit-by-digit: "three zero zero" -> 300, "one five o" -> 150, "two o five" -> 205 * - Bare numbers (no unit): "삼백", "250", "three hundred" */ object VolumeParser { /** Result of a successful volume parse. */ data class ParseResult( val volumeMl: Int, val rawText: String, ) // ── Korean number mappings (DO NOT translate these Korean keys) ────── private val DIGIT_MAP = mapOf( "영" to 0, "공" to 0, "일" to 1, "하나" to 1, "한" to 1, "이" to 2, "둘" to 2, "두" to 2, "삼" to 3, "셋" to 3, "세" to 3, "사" to 4, "넷" to 4, "네" to 4, "오" to 5, "다섯" to 5, "육" to 6, "여섯" to 6, "칠" to 7, "일곱" to 7, "팔" to 8, "여덟" to 8, "구" to 9, "아홉" to 9, ) private val PLACE_MAP = mapOf( "십" to 10, "백" to 100, "천" to 1000, ) private val UNIT_KEYWORDS = listOf( "milliliters", "milliliter", "millimeters", "millimeter", "밀리리터", "밀리미터", "미리리터", "미리미터", "미리", "밀리", "ml", "ML", "mL", ) // ── English number mappings ───────────────────────────────────────── private val EN_ONES = mapOf( "zero" to 0, "one" to 1, "two" to 2, "three" to 3, "four" to 4, "five" to 5, "six" to 6, "seven" to 7, "eight" to 8, "nine" to 9, "ten" to 10, "eleven" to 11, "twelve" to 12, "thirteen" to 13, "fourteen" to 14, "fifteen" to 15, "sixteen" to 16, "seventeen" to 17, "eighteen" to 18, "nineteen" to 19, ) private val EN_TENS = mapOf( "twenty" to 20, "thirty" to 30, "forty" to 40, "fifty" to 50, "sixty" to 60, "seventy" to 70, "eighty" to 80, "ninety" to 90, ) /** Single-digit mapping for digit-by-digit reading (includes "o"/"oh" as zero). */ private val EN_SINGLE_DIGIT = mapOf( "zero" to 0, "o" to 0, "oh" to 0, "one" to 1, "two" to 2, "three" to 3, "four" to 4, "five" to 5, "six" to 6, "seven" to 7, "eight" to 8, "nine" to 9, ) // ── Public API ────────────────────────────────────────────────────── /** * Attempt to extract a volume (1..1000 ml) from [text]. * * @return [ParseResult] when a valid volume is found, `null` otherwise. */ fun parse(text: String): ParseResult? { val cleaned = text.trim() if (cleaned.isEmpty()) return null // 1) Arabic digits (most reliable) extractArabicNumber(cleaned)?.takeIf { it in VALID_RANGE }?.let { return ParseResult(it, cleaned) } // 2) Korean numerals (checked before English to avoid "오" / "o" confusion) extractKoreanNumber(cleaned)?.takeIf { it in VALID_RANGE }?.let { return ParseResult(it, cleaned) } // 3) English compound numerals ("one hundred fifty") extractEnglishNumber(cleaned)?.takeIf { it in VALID_RANGE }?.let { return ParseResult(it, cleaned) } // 4) English digit-by-digit ("three zero zero") -- last because "o" maps to 0 extractDigitByDigitEnglish(cleaned)?.takeIf { it in VALID_RANGE }?.let { return ParseResult(it, cleaned) } return null } /** * Try each [candidates] in order and return the first successful parse. * * Useful because [android.speech.SpeechRecognizer] may return multiple hypotheses. */ fun parseBest(candidates: List): ParseResult? = candidates.firstNotNullOfOrNull { parse(it) } // ── Extraction strategies ─────────────────────────────────────────── private fun extractArabicNumber(text: String): Int? { val stripped = removeUnits(text).replace(Regex("[^0-9]"), "") return stripped.toIntOrNull() } private fun extractKoreanNumber(text: String): Int? { val stripped = removeUnits(text) .replace(Regex("[0-9mlML\\s]", RegexOption.IGNORE_CASE), "") .trim() if (stripped.isEmpty()) return null return parseKoreanNumberString(stripped) } /** * Convert a Korean numeral string to an integer. * Example: "백오십" -> 150, "이백삼" -> 203, "삼백" -> 300 */ private fun parseKoreanNumberString(text: String): Int? { var result = 0 var current = 0 var i = 0 val sortedDigits = DIGIT_MAP.entries.sortedByDescending { it.key.length } while (i < text.length) { // Try place-value keywords first (십, 백, 천) val placeMatch = PLACE_MAP.entries.firstOrNull { text.startsWith(it.key, i) } if (placeMatch != null) { if (current == 0) current = 1 result += current * placeMatch.value current = 0 i += placeMatch.key.length continue } // Try digit words (longest match first: "다섯" before "다") val digitMatch = sortedDigits.firstOrNull { text.startsWith(it.key, i) } if (digitMatch != null) { current = digitMatch.value i += digitMatch.key.length continue } // Unrecognized character -- skip i++ } result += current return result.takeIf { it > 0 } } /** * Convert an English numeral phrase to an integer. * Example: "one hundred fifty" -> 150, "two hundred" -> 200 */ private fun extractEnglishNumber(text: String): Int? { val words = tokenize(text) if (words.isEmpty()) return null var result = 0 var current = 0 for (word in words) { when { word == "thousand" -> { if (current == 0) current = 1 current *= 1_000 } word == "hundred" -> { if (current == 0) current = 1 current *= 100 } word == "and" -> continue word in EN_ONES -> current += EN_ONES.getValue(word) word in EN_TENS -> current += EN_TENS.getValue(word) else -> { // Handle hyphenated forms: "twenty-five" val parts = word.split("-") if (parts.size == 2) { val tens = EN_TENS[parts[0]] val ones = EN_ONES[parts[1]] if (tens != null && ones != null) current += tens + ones } } } } result += current return result.takeIf { it > 0 } } /** * Convert digit-by-digit English reading to an integer. * Example: "three zero zero" -> 300, "one five o" -> 150 * Supports "double" / "triple" modifiers. */ private fun extractDigitByDigitEnglish(text: String): Int? { val words = tokenize(text) if (words.isEmpty()) return null // Expand "double"/"triple" modifiers val expanded = buildList { var i = 0 while (i < words.size) { when { words[i] == "double" && i + 1 < words.size -> { repeat(2) { add(words[i + 1]) } i += 2 } words[i] == "triple" && i + 1 < words.size -> { repeat(3) { add(words[i + 1]) } i += 2 } else -> { add(words[i]) i++ } } } } // Every token must be a single digit val digits = expanded.map { EN_SINGLE_DIGIT[it] ?: return null } if (digits.size !in 2..4) return null val result = digits.fold(0) { acc, d -> acc * 10 + d } return result.takeIf { it > 0 } } // ── Utilities ─────────────────────────────────────────────────────── private fun removeUnits(text: String): String = UNIT_KEYWORDS.fold(text) { acc, unit -> acc.replace(unit, "", ignoreCase = true) }.trim() /** Lowercase, strip units and non-alpha characters, then split on whitespace. */ private fun tokenize(text: String): List { val stripped = removeUnits(text).lowercase().replace(Regex("[^a-z\\s-]"), "").trim() return stripped.split(Regex("\\s+")).filter { it.isNotEmpty() } } private val VALID_RANGE = 1..1000 }