Integrate voiding diary into Piezo catheterization flow

Catheterize button now opens a voiding record sheet with 3 input modes:
- Manual: number TextField (1-1000ml)
- Voice: SpeechRecognizer + VolumeParser (Korean/English)
- Camera: placeholder for YOLO cup measurement (TODO)

On save: "배뇨일지에 등록되었습니다" toast, bladder level resets to 0

New modules (from Uridiary):
- measure/: SimpleMeasureService, YoloDetector, CCPosition, VoidingRecord
- speech/: SpeechRecognizerManager, VolumeParser, HfVolumeExtractor
- assets/urinecup_best.onnx: YOLO model for cup detection

VoidingRecordStore: local SharedPreferences+JSON storage for voiding records

Dependencies added: ONNX Runtime, ML Kit OCR, CameraX, OkHttp
Permissions added: CAMERA, RECORD_AUDIO, INTERNET

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-04-02 11:47:59 +09:00
parent 064f1d1f8d
commit c15fcfccca
12 changed files with 1408 additions and 3 deletions
@@ -0,0 +1,298 @@
/*
* Copyright 2026 Charles KWON (KWON Ohjun)
* charleskwon@medithings.co.kr / charleskwonohjun@gmail.com
* MEDiThings Inc.
* All rights reserved.
*/
package com.example.medilightv2android.speech
import android.util.Log
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.withContext
import okhttp3.MediaType.Companion.toMediaType
import okhttp3.OkHttpClient
import okhttp3.Request
import okhttp3.RequestBody.Companion.toRequestBody
import org.json.JSONArray
import org.json.JSONObject
import java.util.concurrent.TimeUnit
/**
* Extracts a milliliter volume from speech-recognition text via the
* Hugging Face Inference API (Qwen 2.5-72B model).
*
* Selects a language-specific system prompt to maximize extraction accuracy.
* Falls back to [VolumeParser] when the API key is missing or the request fails.
*
* @param apiKey Hugging Face bearer token.
*/
class HfVolumeExtractor(private val apiKey: String) {
/** Successful extraction result. */
data class ExtractResult(
val volumeMl: Int,
val rawText: String,
val explanation: String,
)
private val client = OkHttpClient.Builder()
.connectTimeout(CONNECT_TIMEOUT_SECONDS, TimeUnit.SECONDS)
.readTimeout(READ_TIMEOUT_SECONDS, TimeUnit.SECONDS)
.build()
/**
* Send [speechText] to the AI model using a prompt tailored to [language],
* and return the extracted volume, or `null` when extraction fails.
*/
suspend fun extract(
speechText: String,
language: SpeechLanguage = SpeechLanguage.KOREAN,
): ExtractResult? = withContext(Dispatchers.IO) {
Log.d(TAG, "Starting extraction | lang=${language.code} | input: $speechText")
if (apiKey.isBlank()) {
Log.w(TAG, "API key is blank; falling back to local parser")
return@withContext fallbackParse(speechText)
}
try {
val prompt = getSystemPrompt(language)
val requestBody = buildRequestBody(speechText, prompt)
val request = Request.Builder()
.url(API_URL)
.addHeader("Authorization", "Bearer $apiKey")
.addHeader("Content-Type", JSON_MEDIA_TYPE)
.post(requestBody.toRequestBody(JSON_MEDIA_TYPE.toMediaType()))
.build()
val response = client.newCall(request).execute()
val body = response.body?.string() ?: return@withContext null
Log.d(TAG, "Response code: ${response.code}")
if (!response.isSuccessful) {
Log.e(TAG, "API failed (${response.code}); falling back to local parser")
return@withContext fallbackParse(speechText)
}
parseApiResponse(body, speechText)
} catch (e: Exception) {
Log.e(TAG, "API error: ${e.message}", e)
fallbackParse(speechText)
}
}
// ── Private helpers ─────────────────────────────────────────────────
private fun buildRequestBody(speechText: String, systemPrompt: String): String =
JSONObject().apply {
put("model", MODEL_ID)
put("max_tokens", MAX_TOKENS)
put("stream", false)
put("messages", JSONArray().apply {
put(JSONObject().apply {
put("role", "system")
put("content", systemPrompt)
})
put(JSONObject().apply {
put("role", "user")
put("content", speechText)
})
})
}.toString()
private fun parseApiResponse(body: String, speechText: String): ExtractResult? {
val content = JSONObject(body)
.getJSONArray("choices")
.getJSONObject(0)
.getJSONObject("message")
.getString("content")
Log.d(TAG, "Model response: $content")
val jsonStr = extractJsonObject(content)
val resultJson = JSONObject(jsonStr)
val ml = resultJson.getInt("ml")
val explanation = resultJson.optString("explanation", "")
Log.d(TAG, "Parsed result: ${ml}ml - $explanation")
return if (ml in VALID_RANGE) {
ExtractResult(ml, speechText, explanation)
} else {
Log.w(TAG, "Volume out of range: ${ml}ml")
null
}
}
private fun extractJsonObject(text: String): String {
val start = text.indexOf('{')
val end = text.lastIndexOf('}')
return if (start >= 0 && end > start) text.substring(start, end + 1) else text
}
private fun fallbackParse(text: String): ExtractResult? {
val result = VolumeParser.parse(text) ?: return null
Log.d(TAG, "Local parse result: ${result.volumeMl}ml")
return ExtractResult(result.volumeMl, result.rawText, "Local parse")
}
/** Return a system prompt optimized for the given [language]. */
private fun getSystemPrompt(language: SpeechLanguage): String = when (language) {
SpeechLanguage.KOREAN -> PROMPT_KOREAN
SpeechLanguage.ENGLISH -> PROMPT_ENGLISH
SpeechLanguage.CHINESE -> PROMPT_CHINESE
SpeechLanguage.JAPANESE -> PROMPT_JAPANESE
}
companion object {
private const val TAG = "HfVolumeExtractor"
private const val API_URL = "https://router.huggingface.co/v1/chat/completions"
private const val MODEL_ID = "Qwen/Qwen2.5-72B-Instruct"
private const val JSON_MEDIA_TYPE = "application/json"
private const val MAX_TOKENS = 100
private const val CONNECT_TIMEOUT_SECONDS = 15L
private const val READ_TIMEOUT_SECONDS = 30L
private val VALID_RANGE = 1..1000
// ── Language-specific prompts ───────────────────────────────────
private val PROMPT_BASE = """
## Your Mission
You are specialized in extracting MILLILITER (ml) volumes from voice input.
The text you receive is raw speech-to-text output from a medical/lab device.
Your ONLY job: find the ml number the user intended to say.
## Rules
- Extract a single integer in ml, range 1~1000
- mm and ml are the SAME in this context (user ALWAYS means ml, never millimeters)
- cc and ml are the SAME (1cc = 1ml)
- If ambiguous, prefer the most likely medical volume
- Look at ALL candidates (separated by " / ") to find consensus on the number
- If no valid volume found, return -1
- Respond ONLY in this JSON format:
""".trimIndent()
private val PROMPT_KOREAN = """
You are a milliliter (ml) volume extraction AI.
## Context
This is a KOREAN speech-to-text result from a medical/lab voice input device.
The user spoke in KOREAN to input a liquid volume in milliliters (ml).
Your task: understand Korean speech and extract the ml value accurately.
## Input Format
Multiple STT candidates separated by " / " (e.g. "95mm / 95ml / 95 mm").
STT often confuses mm/ml/미리/밀리 — the user ALWAYS means milliliters (ml).
## Korean-Specific Patterns
- Pure Korean numerals: "삼백 밀리", "백오십 미리"
- Numbers with units: "300ml", "150 미리"
- Korean-English mix: "삼백 ml", "원헌드레드 미리"
- Misheard Korean that is actually English: "쓰리헌드레드", "투헌드레드 피프티"
- Phonetic Korean of English: "파이브 헌드레드" = 500, "투 피프티" = 250
- Casual/slang: "반 리터" = 500, "한 컵" = 250, "반" = 500, "쿼터" = 250
- Unit variations: 밀리리터, 밀리미터, 미리, 밀리, ml, cc, 시시
## Critical: Korean Phonetic Analysis
Korean STT may transcribe English words as Korean phonetics.
SOUND THEM OUT to decode:
- 쓰리 = three, 투 = two, 원 = one, 포 = four, 파이브 = five
- 식스 = six, 세븐 = seven, 에잇 = eight, 나인 = nine, 텐 = ten
- 헌드레드 = hundred, 피프티 = fifty, 트웬티 = twenty, 서티 = thirty
- 지로/제로 = zero, 오 = five OR "o" (context dependent)
- 더블 = double, 트리플 = triple
$PROMPT_BASE
{"ml": <integer>, "explanation": "<한국어로 간단한 설명>"}
""".trimIndent()
private val PROMPT_ENGLISH = """
You are a milliliter (ml) volume extraction AI.
## Context
This is an ENGLISH speech-to-text result from a medical/lab voice input device.
The user spoke in ENGLISH to input a liquid volume in milliliters (ml).
Your task: understand English speech and extract the ml value accurately.
## Input Format
Multiple STT candidates separated by " / " (e.g. "95mm / 95ml / 95 mm").
STT often confuses mm/ml — the user ALWAYS means milliliters (ml).
## English-Specific Patterns
- Standard: "three hundred milliliters", "one fifty ml"
- Digit-by-digit: "three zero zero" = 300, "one five o" = 150, "two o five" = 205
- Casual: "half a liter" = 500, "a quarter liter" = 250, "half" = 500
- Compound: "one hundred and fifty", "two fifty"
- With modifiers: "double zero" = 00, "triple zero" = 000
- Unit variations: ml, milliliters, millimeters, cc
$PROMPT_BASE
{"ml": <integer>, "explanation": "<brief reason in English>"}
""".trimIndent()
private val PROMPT_CHINESE = """
你是一个毫升(ml)容量提取AI。
## 背景
这是来自医疗/实验室语音输入设备的中文语音识别结果。
用户用中文说出了一个液体容量(毫升/ml)。
你的任务:准确理解中文语音并提取毫升值。
## 输入格式
多个语音识别候选项用 " / " 分隔(例如 "95毫升 / 95ml / 九十五")。
语音识别经常混淆毫米/毫升——用户始终指的是毫升(ml)。
## 中文特有模式
- 中文数字: "三百毫升", "一百五十", "二百五"
- 阿拉伯数字+中文单位: "300毫升", "150ml"
- 口语: "半升" = 500, "一杯" = 250, "半" = 500
- 口语简称: "二百五" = 250(此处不是骂人), "三百" = 300
- 单位变体: 毫升, 毫米, ml, cc, 西西
## 关键:中文数字规则
- 一百五 = 150(一百五十的简称)
- 二百五 = 250(二百五十的简称)
- 两百 = 200 (两 = 2 in spoken Chinese)
- 半升 = 500ml, 四分之一升 = 250ml
- 一千 = 1000
$PROMPT_BASE
{"ml": <integer>, "explanation": "<用中文简要说明>"}
""".trimIndent()
private val PROMPT_JAPANESE = """
あなたはミリリットル(ml)容量抽出AIです。
## 背景
これは医療・実験室の音声入力デバイスからの日本語音声認識結果です。
ユーザーは日本語で液体の容量(ミリリットル/ml)を発話しました。
あなたのタスク:日本語の音声を正確に理解し、ml値を抽出すること。
## 入力形式
複数の音声認識候補が " / " で区切られています(例:"95ミリ / 95ml / きゅうじゅうご")。
音声認識はミリメートル/ミリリットルを混同することが多い — ユーザーは常にミリリットル(ml)を意味します。
## 日本語特有のパターン
- 日本語数字: "さんびゃくミリ", "ひゃくごじゅう"
- 漢数字: "三百ミリリットル", "百五十"
- 数字+単位: "300ml", "150ミリ"
- カジュアル: "半リットル" = 500, "コップ一杯" = 250
- 省略形: "にひゃくごじゅう" = 250, "さんびゃく" = 300
- 単位バリエーション: ミリリットル, ミリ, ml, cc, シーシー
## 重要:日本語数字ルール
- ひゃくごじゅう = 150
- にひゃく = 200
- にひゃくご = 250(にひゃくごじゅうの略)
- さんびゃく = 300
- ごひゃく = 500
- せん = 1000
- 半分 = half (context: 半リットル = 500ml)
$PROMPT_BASE
{"ml": <integer>, "explanation": "<日本語で簡単な説明>"}
""".trimIndent()
}
}
@@ -0,0 +1,103 @@
/*
* Copyright 2026 Charles KWON (KWON Ohjun)
* charleskwon@medithings.co.kr / charleskwonohjun@gmail.com
* MEDiThings Inc.
* All rights reserved.
*/
package com.example.medilightv2android.speech
import android.content.Context
import android.content.Intent
import android.os.Bundle
import android.speech.RecognitionListener
import android.speech.RecognizerIntent
import android.speech.SpeechRecognizer
/**
* Supported speech-recognition locales.
*
* @property code BCP-47 language tag used by [RecognizerIntent].
* @property label Human-readable name shown in the UI language toggle.
*/
enum class SpeechLanguage(val code: String, val label: String) {
KOREAN("ko-KR", "한국어"),
ENGLISH("en-US", "English"),
CHINESE("zh-CN", "中文"),
JAPANESE("ja-JP", "日本語"),
}
/**
* Thin wrapper around Android's [SpeechRecognizer] that manages the recognizer lifecycle
* and exposes results through simple callbacks.
*
* @param context Application or Activity context.
* @param language Target speech-recognition locale.
* @param onResult Invoked with the ranked list of recognition candidates.
* @param onError Invoked with a [SpeechRecognizer] error code on failure.
* @param onReady Invoked when the recognizer is ready to accept speech.
*/
class SpeechRecognizerManager(
private val context: Context,
private val language: SpeechLanguage = SpeechLanguage.KOREAN,
private val onResult: (List<String>) -> Unit,
private val onError: (Int) -> Unit,
private val onReady: () -> Unit,
) {
private var speechRecognizer: SpeechRecognizer? = null
private val recognitionIntent: Intent by lazy {
Intent(RecognizerIntent.ACTION_RECOGNIZE_SPEECH).apply {
putExtra(RecognizerIntent.EXTRA_LANGUAGE_MODEL, RecognizerIntent.LANGUAGE_MODEL_FREE_FORM)
putExtra(RecognizerIntent.EXTRA_LANGUAGE, language.code)
putExtra(RecognizerIntent.EXTRA_LANGUAGE_PREFERENCE, language.code)
putExtra(RecognizerIntent.EXTRA_ONLY_RETURN_LANGUAGE_PREFERENCE, false)
putExtra(RecognizerIntent.EXTRA_MAX_RESULTS, MAX_RESULTS)
putExtra(RecognizerIntent.EXTRA_PARTIAL_RESULTS, true)
}
}
/** Create a fresh [SpeechRecognizer] and begin listening. */
fun startListening() {
destroy()
speechRecognizer = SpeechRecognizer.createSpeechRecognizer(context).apply {
setRecognitionListener(createListener())
startListening(recognitionIntent)
}
}
/** Ask the recognizer to stop capturing audio. */
fun stopListening() {
speechRecognizer?.stopListening()
}
/** Release the underlying [SpeechRecognizer] resources. */
fun destroy() {
speechRecognizer?.destroy()
speechRecognizer = null
}
private fun createListener(): RecognitionListener = object : RecognitionListener {
override fun onReadyForSpeech(params: Bundle?) { onReady() }
override fun onBeginningOfSpeech() {}
override fun onRmsChanged(rmsdB: Float) {}
override fun onBufferReceived(buffer: ByteArray?) {}
override fun onEndOfSpeech() {}
override fun onError(error: Int) { this@SpeechRecognizerManager.onError(error) }
override fun onResults(results: Bundle?) {
val matches = results
?.getStringArrayList(SpeechRecognizer.RESULTS_RECOGNITION)
.orEmpty()
onResult(matches)
}
override fun onPartialResults(partialResults: Bundle?) {}
override fun onEvent(eventType: Int, params: Bundle?) {}
}
companion object {
private const val MAX_RESULTS = 5
}
}
@@ -0,0 +1,260 @@
/*
* Copyright 2026 Charles KWON (KWON Ohjun)
* charleskwon@medithings.co.kr / charleskwonohjun@gmail.com
* MEDiThings Inc.
* All rights reserved.
*/
package com.example.medilightv2android.speech
/**
* Local (offline) parser that extracts a volume in milliliters from speech-recognition text.
*
* Supported patterns:
* - Arabic digits with unit: "150ml", "200 미리"
* - Korean numerals with unit: "백오십 밀리리터", "이백 미리"
* - English numerals with unit: "one hundred fifty ml", "two hundred milliliters"
* - English digit-by-digit: "three zero zero" -> 300, "one five o" -> 150, "two o five" -> 205
* - Bare numbers (no unit): "삼백", "250", "three hundred"
*/
object VolumeParser {
/** Result of a successful volume parse. */
data class ParseResult(
val volumeMl: Int,
val rawText: String,
)
// ── Korean number mappings (DO NOT translate these Korean keys) ──────
private val DIGIT_MAP = mapOf(
"영" to 0, "공" to 0,
"일" to 1, "하나" to 1, "한" to 1,
"이" to 2, "둘" to 2, "두" to 2,
"삼" to 3, "셋" to 3, "세" to 3,
"사" to 4, "넷" to 4, "네" to 4,
"오" to 5, "다섯" to 5,
"육" to 6, "여섯" to 6,
"칠" to 7, "일곱" to 7,
"팔" to 8, "여덟" to 8,
"구" to 9, "아홉" to 9,
)
private val PLACE_MAP = mapOf(
"십" to 10,
"백" to 100,
"천" to 1000,
)
private val UNIT_KEYWORDS = listOf(
"milliliters", "milliliter", "millimeters", "millimeter",
"밀리리터", "밀리미터", "미리리터", "미리미터",
"미리", "밀리", "ml", "ML", "mL",
)
// ── English number mappings ─────────────────────────────────────────
private val EN_ONES = mapOf(
"zero" to 0, "one" to 1, "two" to 2, "three" to 3, "four" to 4,
"five" to 5, "six" to 6, "seven" to 7, "eight" to 8, "nine" to 9,
"ten" to 10, "eleven" to 11, "twelve" to 12, "thirteen" to 13,
"fourteen" to 14, "fifteen" to 15, "sixteen" to 16, "seventeen" to 17,
"eighteen" to 18, "nineteen" to 19,
)
private val EN_TENS = mapOf(
"twenty" to 20, "thirty" to 30, "forty" to 40, "fifty" to 50,
"sixty" to 60, "seventy" to 70, "eighty" to 80, "ninety" to 90,
)
/** Single-digit mapping for digit-by-digit reading (includes "o"/"oh" as zero). */
private val EN_SINGLE_DIGIT = mapOf(
"zero" to 0, "o" to 0, "oh" to 0,
"one" to 1, "two" to 2, "three" to 3, "four" to 4,
"five" to 5, "six" to 6, "seven" to 7, "eight" to 8, "nine" to 9,
)
// ── Public API ──────────────────────────────────────────────────────
/**
* Attempt to extract a volume (1..1000 ml) from [text].
*
* @return [ParseResult] when a valid volume is found, `null` otherwise.
*/
fun parse(text: String): ParseResult? {
val cleaned = text.trim()
if (cleaned.isEmpty()) return null
// 1) Arabic digits (most reliable)
extractArabicNumber(cleaned)?.takeIf { it in VALID_RANGE }?.let {
return ParseResult(it, cleaned)
}
// 2) Korean numerals (checked before English to avoid "오" / "o" confusion)
extractKoreanNumber(cleaned)?.takeIf { it in VALID_RANGE }?.let {
return ParseResult(it, cleaned)
}
// 3) English compound numerals ("one hundred fifty")
extractEnglishNumber(cleaned)?.takeIf { it in VALID_RANGE }?.let {
return ParseResult(it, cleaned)
}
// 4) English digit-by-digit ("three zero zero") -- last because "o" maps to 0
extractDigitByDigitEnglish(cleaned)?.takeIf { it in VALID_RANGE }?.let {
return ParseResult(it, cleaned)
}
return null
}
/**
* Try each [candidates] in order and return the first successful parse.
*
* Useful because [android.speech.SpeechRecognizer] may return multiple hypotheses.
*/
fun parseBest(candidates: List<String>): ParseResult? =
candidates.firstNotNullOfOrNull { parse(it) }
// ── Extraction strategies ───────────────────────────────────────────
private fun extractArabicNumber(text: String): Int? {
val stripped = removeUnits(text).replace(Regex("[^0-9]"), "")
return stripped.toIntOrNull()
}
private fun extractKoreanNumber(text: String): Int? {
val stripped = removeUnits(text)
.replace(Regex("[0-9mlML\\s]", RegexOption.IGNORE_CASE), "")
.trim()
if (stripped.isEmpty()) return null
return parseKoreanNumberString(stripped)
}
/**
* Convert a Korean numeral string to an integer.
* Example: "백오십" -> 150, "이백삼" -> 203, "삼백" -> 300
*/
private fun parseKoreanNumberString(text: String): Int? {
var result = 0
var current = 0
var i = 0
val sortedDigits = DIGIT_MAP.entries.sortedByDescending { it.key.length }
while (i < text.length) {
// Try place-value keywords first (십, 백, 천)
val placeMatch = PLACE_MAP.entries.firstOrNull { text.startsWith(it.key, i) }
if (placeMatch != null) {
if (current == 0) current = 1
result += current * placeMatch.value
current = 0
i += placeMatch.key.length
continue
}
// Try digit words (longest match first: "다섯" before "다")
val digitMatch = sortedDigits.firstOrNull { text.startsWith(it.key, i) }
if (digitMatch != null) {
current = digitMatch.value
i += digitMatch.key.length
continue
}
// Unrecognized character -- skip
i++
}
result += current
return result.takeIf { it > 0 }
}
/**
* Convert an English numeral phrase to an integer.
* Example: "one hundred fifty" -> 150, "two hundred" -> 200
*/
private fun extractEnglishNumber(text: String): Int? {
val words = tokenize(text)
if (words.isEmpty()) return null
var result = 0
var current = 0
for (word in words) {
when {
word == "thousand" -> {
if (current == 0) current = 1
current *= 1_000
}
word == "hundred" -> {
if (current == 0) current = 1
current *= 100
}
word == "and" -> continue
word in EN_ONES -> current += EN_ONES.getValue(word)
word in EN_TENS -> current += EN_TENS.getValue(word)
else -> {
// Handle hyphenated forms: "twenty-five"
val parts = word.split("-")
if (parts.size == 2) {
val tens = EN_TENS[parts[0]]
val ones = EN_ONES[parts[1]]
if (tens != null && ones != null) current += tens + ones
}
}
}
}
result += current
return result.takeIf { it > 0 }
}
/**
* Convert digit-by-digit English reading to an integer.
* Example: "three zero zero" -> 300, "one five o" -> 150
* Supports "double" / "triple" modifiers.
*/
private fun extractDigitByDigitEnglish(text: String): Int? {
val words = tokenize(text)
if (words.isEmpty()) return null
// Expand "double"/"triple" modifiers
val expanded = buildList {
var i = 0
while (i < words.size) {
when {
words[i] == "double" && i + 1 < words.size -> {
repeat(2) { add(words[i + 1]) }
i += 2
}
words[i] == "triple" && i + 1 < words.size -> {
repeat(3) { add(words[i + 1]) }
i += 2
}
else -> {
add(words[i])
i++
}
}
}
}
// Every token must be a single digit
val digits = expanded.map { EN_SINGLE_DIGIT[it] ?: return null }
if (digits.size !in 2..4) return null
val result = digits.fold(0) { acc, d -> acc * 10 + d }
return result.takeIf { it > 0 }
}
// ── Utilities ───────────────────────────────────────────────────────
private fun removeUnits(text: String): String =
UNIT_KEYWORDS.fold(text) { acc, unit -> acc.replace(unit, "", ignoreCase = true) }.trim()
/** Lowercase, strip units and non-alpha characters, then split on whitespace. */
private fun tokenize(text: String): List<String> {
val stripped = removeUnits(text).lowercase().replace(Regex("[^a-z\\s-]"), "").trim()
return stripped.split(Regex("\\s+")).filter { it.isNotEmpty() }
}
private val VALID_RANGE = 1..1000
}