diff --git a/app/src/main/java/com/medithings/vesiscan/speech/HfVolumeExtractor.kt b/app/src/main/java/com/medithings/vesiscan/speech/HfVolumeExtractor.kt deleted file mode 100644 index 72c53c3..0000000 --- a/app/src/main/java/com/medithings/vesiscan/speech/HfVolumeExtractor.kt +++ /dev/null @@ -1,298 +0,0 @@ -/* - * Copyright 2026 Charles KWON (KWON Ohjun) - * charleskwon@medithings.co.kr / charleskwonohjun@gmail.com - * MEDiThings Inc. - * All rights reserved. - */ -package com.medithings.vesiscan.speech - -import android.util.Log -import kotlinx.coroutines.Dispatchers -import kotlinx.coroutines.withContext -import okhttp3.MediaType.Companion.toMediaType -import okhttp3.OkHttpClient -import okhttp3.Request -import okhttp3.RequestBody.Companion.toRequestBody -import org.json.JSONArray -import org.json.JSONObject -import java.util.concurrent.TimeUnit - -/** - * Extracts a milliliter volume from speech-recognition text via the - * Hugging Face Inference API (Qwen 2.5-72B model). - * - * Selects a language-specific system prompt to maximize extraction accuracy. - * Falls back to [VolumeParser] when the API key is missing or the request fails. - * - * @param apiKey Hugging Face bearer token. - */ -class HfVolumeExtractor(private val apiKey: String) { - - /** Successful extraction result. */ - data class ExtractResult( - val volumeMl: Int, - val rawText: String, - val explanation: String, - ) - - private val client = OkHttpClient.Builder() - .connectTimeout(CONNECT_TIMEOUT_SECONDS, TimeUnit.SECONDS) - .readTimeout(READ_TIMEOUT_SECONDS, TimeUnit.SECONDS) - .build() - - /** - * Send [speechText] to the AI model using a prompt tailored to [language], - * and return the extracted volume, or `null` when extraction fails. - */ - suspend fun extract( - speechText: String, - language: SpeechLanguage = SpeechLanguage.KOREAN, - ): ExtractResult? = withContext(Dispatchers.IO) { - Log.d(TAG, "Starting extraction | lang=${language.code} | input: $speechText") - - if (apiKey.isBlank()) { - Log.w(TAG, "API key is blank; falling back to local parser") - return@withContext fallbackParse(speechText) - } - - try { - val prompt = getSystemPrompt(language) - val requestBody = buildRequestBody(speechText, prompt) - - val request = Request.Builder() - .url(API_URL) - .addHeader("Authorization", "Bearer $apiKey") - .addHeader("Content-Type", JSON_MEDIA_TYPE) - .post(requestBody.toRequestBody(JSON_MEDIA_TYPE.toMediaType())) - .build() - - val response = client.newCall(request).execute() - val body = response.body?.string() ?: return@withContext null - - Log.d(TAG, "Response code: ${response.code}") - - if (!response.isSuccessful) { - Log.e(TAG, "API failed (${response.code}); falling back to local parser") - return@withContext fallbackParse(speechText) - } - - parseApiResponse(body, speechText) - } catch (e: Exception) { - Log.e(TAG, "API error: ${e.message}", e) - fallbackParse(speechText) - } - } - - // ── Private helpers ───────────────────────────────────────────────── - - private fun buildRequestBody(speechText: String, systemPrompt: String): String = - JSONObject().apply { - put("model", MODEL_ID) - put("max_tokens", MAX_TOKENS) - put("stream", false) - put("messages", JSONArray().apply { - put(JSONObject().apply { - put("role", "system") - put("content", systemPrompt) - }) - put(JSONObject().apply { - put("role", "user") - put("content", speechText) - }) - }) - }.toString() - - private fun parseApiResponse(body: String, speechText: String): ExtractResult? { - val content = JSONObject(body) - .getJSONArray("choices") - .getJSONObject(0) - .getJSONObject("message") - .getString("content") - - Log.d(TAG, "Model response: $content") - - val jsonStr = extractJsonObject(content) - val resultJson = JSONObject(jsonStr) - val ml = resultJson.getInt("ml") - val explanation = resultJson.optString("explanation", "") - - Log.d(TAG, "Parsed result: ${ml}ml - $explanation") - - return if (ml in VALID_RANGE) { - ExtractResult(ml, speechText, explanation) - } else { - Log.w(TAG, "Volume out of range: ${ml}ml") - null - } - } - - private fun extractJsonObject(text: String): String { - val start = text.indexOf('{') - val end = text.lastIndexOf('}') - return if (start >= 0 && end > start) text.substring(start, end + 1) else text - } - - private fun fallbackParse(text: String): ExtractResult? { - val result = VolumeParser.parse(text) ?: return null - Log.d(TAG, "Local parse result: ${result.volumeMl}ml") - return ExtractResult(result.volumeMl, result.rawText, "Local parse") - } - - /** Return a system prompt optimized for the given [language]. */ - private fun getSystemPrompt(language: SpeechLanguage): String = when (language) { - SpeechLanguage.KOREAN -> PROMPT_KOREAN - SpeechLanguage.ENGLISH -> PROMPT_ENGLISH - SpeechLanguage.CHINESE -> PROMPT_CHINESE - SpeechLanguage.JAPANESE -> PROMPT_JAPANESE - } - - companion object { - private const val TAG = "HfVolumeExtractor" - private const val API_URL = "https://router.huggingface.co/v1/chat/completions" - private const val MODEL_ID = "Qwen/Qwen2.5-72B-Instruct" - private const val JSON_MEDIA_TYPE = "application/json" - private const val MAX_TOKENS = 100 - private const val CONNECT_TIMEOUT_SECONDS = 15L - private const val READ_TIMEOUT_SECONDS = 30L - private val VALID_RANGE = 1..1000 - - // ── Language-specific prompts ─────────────────────────────────── - - private val PROMPT_BASE = """ - ## Your Mission - You are specialized in extracting MILLILITER (ml) volumes from voice input. - The text you receive is raw speech-to-text output from a medical/lab device. - Your ONLY job: find the ml number the user intended to say. - - ## Rules - - Extract a single integer in ml, range 1~1000 - - mm and ml are the SAME in this context (user ALWAYS means ml, never millimeters) - - cc and ml are the SAME (1cc = 1ml) - - If ambiguous, prefer the most likely medical volume - - Look at ALL candidates (separated by " / ") to find consensus on the number - - If no valid volume found, return -1 - - Respond ONLY in this JSON format: - """.trimIndent() - - private val PROMPT_KOREAN = """ - You are a milliliter (ml) volume extraction AI. - - ## Context - This is a KOREAN speech-to-text result from a medical/lab voice input device. - The user spoke in KOREAN to input a liquid volume in milliliters (ml). - Your task: understand Korean speech and extract the ml value accurately. - - ## Input Format - Multiple STT candidates separated by " / " (e.g. "95mm / 95ml / 95 mm"). - STT often confuses mm/ml/미리/밀리 — the user ALWAYS means milliliters (ml). - - ## Korean-Specific Patterns - - Pure Korean numerals: "삼백 밀리", "백오십 미리" - - Numbers with units: "300ml", "150 미리" - - Korean-English mix: "삼백 ml", "원헌드레드 미리" - - Misheard Korean that is actually English: "쓰리헌드레드", "투헌드레드 피프티" - - Phonetic Korean of English: "파이브 헌드레드" = 500, "투 피프티" = 250 - - Casual/slang: "반 리터" = 500, "한 컵" = 250, "반" = 500, "쿼터" = 250 - - Unit variations: 밀리리터, 밀리미터, 미리, 밀리, ml, cc, 시시 - - ## Critical: Korean Phonetic Analysis - Korean STT may transcribe English words as Korean phonetics. - SOUND THEM OUT to decode: - - 쓰리 = three, 투 = two, 원 = one, 포 = four, 파이브 = five - - 식스 = six, 세븐 = seven, 에잇 = eight, 나인 = nine, 텐 = ten - - 헌드레드 = hundred, 피프티 = fifty, 트웬티 = twenty, 서티 = thirty - - 지로/제로 = zero, 오 = five OR "o" (context dependent) - - 더블 = double, 트리플 = triple - - $PROMPT_BASE - {"ml": , "explanation": "<한국어로 간단한 설명>"} - """.trimIndent() - - private val PROMPT_ENGLISH = """ - You are a milliliter (ml) volume extraction AI. - - ## Context - This is an ENGLISH speech-to-text result from a medical/lab voice input device. - The user spoke in ENGLISH to input a liquid volume in milliliters (ml). - Your task: understand English speech and extract the ml value accurately. - - ## Input Format - Multiple STT candidates separated by " / " (e.g. "95mm / 95ml / 95 mm"). - STT often confuses mm/ml — the user ALWAYS means milliliters (ml). - - ## English-Specific Patterns - - Standard: "three hundred milliliters", "one fifty ml" - - Digit-by-digit: "three zero zero" = 300, "one five o" = 150, "two o five" = 205 - - Casual: "half a liter" = 500, "a quarter liter" = 250, "half" = 500 - - Compound: "one hundred and fifty", "two fifty" - - With modifiers: "double zero" = 00, "triple zero" = 000 - - Unit variations: ml, milliliters, millimeters, cc - - $PROMPT_BASE - {"ml": , "explanation": ""} - """.trimIndent() - - private val PROMPT_CHINESE = """ - 你是一个毫升(ml)容量提取AI。 - - ## 背景 - 这是来自医疗/实验室语音输入设备的中文语音识别结果。 - 用户用中文说出了一个液体容量(毫升/ml)。 - 你的任务:准确理解中文语音并提取毫升值。 - - ## 输入格式 - 多个语音识别候选项用 " / " 分隔(例如 "95毫升 / 95ml / 九十五")。 - 语音识别经常混淆毫米/毫升——用户始终指的是毫升(ml)。 - - ## 中文特有模式 - - 中文数字: "三百毫升", "一百五十", "二百五" - - 阿拉伯数字+中文单位: "300毫升", "150ml" - - 口语: "半升" = 500, "一杯" = 250, "半" = 500 - - 口语简称: "二百五" = 250(此处不是骂人), "三百" = 300 - - 单位变体: 毫升, 毫米, ml, cc, 西西 - - ## 关键:中文数字规则 - - 一百五 = 150(一百五十的简称) - - 二百五 = 250(二百五十的简称) - - 两百 = 200 (两 = 2 in spoken Chinese) - - 半升 = 500ml, 四分之一升 = 250ml - - 一千 = 1000 - - $PROMPT_BASE - {"ml": , "explanation": "<用中文简要说明>"} - """.trimIndent() - - private val PROMPT_JAPANESE = """ - あなたはミリリットル(ml)容量抽出AIです。 - - ## 背景 - これは医療・実験室の音声入力デバイスからの日本語音声認識結果です。 - ユーザーは日本語で液体の容量(ミリリットル/ml)を発話しました。 - あなたのタスク:日本語の音声を正確に理解し、ml値を抽出すること。 - - ## 入力形式 - 複数の音声認識候補が " / " で区切られています(例:"95ミリ / 95ml / きゅうじゅうご")。 - 音声認識はミリメートル/ミリリットルを混同することが多い — ユーザーは常にミリリットル(ml)を意味します。 - - ## 日本語特有のパターン - - 日本語数字: "さんびゃくミリ", "ひゃくごじゅう" - - 漢数字: "三百ミリリットル", "百五十" - - 数字+単位: "300ml", "150ミリ" - - カジュアル: "半リットル" = 500, "コップ一杯" = 250 - - 省略形: "にひゃくごじゅう" = 250, "さんびゃく" = 300 - - 単位バリエーション: ミリリットル, ミリ, ml, cc, シーシー - - ## 重要:日本語数字ルール - - ひゃくごじゅう = 150 - - にひゃく = 200 - - にひゃくご = 250(にひゃくごじゅうの略) - - さんびゃく = 300 - - ごひゃく = 500 - - せん = 1000 - - 半分 = half (context: 半リットル = 500ml) - - $PROMPT_BASE - {"ml": , "explanation": "<日本語で簡単な説明>"} - """.trimIndent() - } -} diff --git a/app/src/main/res/raw/bladdy_loading3.riv b/app/src/main/res/raw/bladdy_loading3.riv deleted file mode 100644 index 6a80901..0000000 Binary files a/app/src/main/res/raw/bladdy_loading3.riv and /dev/null differ