From 252692417acd47baa919e165fda2bd1ac6b11119 Mon Sep 17 00:00:00 2001 From: jjangddu Date: Tue, 30 Jun 2026 17:49:28 +0900 Subject: [PATCH] =?UTF-8?q?tools:=20analyze=5Fcsv.py=20=E2=80=94=20App=20C?= =?UTF-8?q?SV=20=E2=86=92=20Python=20pipeline=20=EB=B9=84=EA=B5=90/?= =?UTF-8?q?=EA=B2=80=EC=A6=9D=20CLI?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit App 측정 시 자동 저장되는 CSV (Downloads/VesiScan_ADC/*.csv) 를 appshare for_app_share / vesiscan_test.library.method_d 로 재계산해 다음을 비교/시각화하는 영구 분석 툴. 4 modes: 1. Basic compare: App BV vs Python BV (mean/std/bias/CV) 2. Ablation: 5 algorithm config (DPS, lr floor 등) 개별 영향 3. Plot: BV timeseries / app↔py scatter / lr_ratio histogram 4. Cycle inspect: 특정 scan_id 의 raw → walls → BV detail 설계 의도: - 알고리즘 변경 후 회귀 자동 검증 - 임상 측정 정확도 (catheter ground truth) 비교 - device-to-device variance 분석 - lr_ratio 분포로 phantom vs human anatomy 패턴 차이 확인 appshare 경로 자동 탐색 + APPSHARE_DIR env var 지원. 검증: phantom 150 mL CSV (2547 scans) 로 smoke test 통과. APP mean=147.6 (-1.6%), Python NEW mean=93.0 (-38%, lr=0.67) → ablation 으로 lr_ratio 가 차이의 주범 확인. Co-Authored-By: Claude Opus 4.7 --- tools/.gitignore | 4 + tools/README.md | 77 +++++++++ tools/analyze_csv.py | 390 +++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 471 insertions(+) create mode 100644 tools/.gitignore create mode 100644 tools/README.md create mode 100644 tools/analyze_csv.py diff --git a/tools/.gitignore b/tools/.gitignore new file mode 100644 index 0000000..1bbed5e --- /dev/null +++ b/tools/.gitignore @@ -0,0 +1,4 @@ +out_smoke/ +out/ +*.png +__pycache__/ diff --git a/tools/README.md b/tools/README.md new file mode 100644 index 0000000..500c546 --- /dev/null +++ b/tools/README.md @@ -0,0 +1,77 @@ +# tools/ + +VesiScan-Basic 측정 결과 분석 / 검증 툴. + +## analyze_csv.py — App CSV → Python pipeline 비교 + +App 이 측정 시 저장하는 CSV (`Downloads/VesiScan_ADC/*.csv`) 를 읽어, +appshare 의 Python pipeline (`for_app_share`) 으로 BV 를 재계산하고 +다음을 비교한다: + +- App-computed BV vs Python BV (per-scan, aggregate) +- Method D wall detection rate +- lr_ratio 분포 +- DPS / lr floor 등 알고리즘 변경 사항의 ablation + +### Setup + +```bash +# 의존성 +pip install numpy pandas matplotlib + +# appshare repo 경로 (둘 중 하나) +export APPSHARE_DIR=/path/to/appshare/piezo-phantom-test +# 또는 명시: --appshare /path/... +``` + +### Usage + +```bash +# 기본 비교 (phantom 150 mL) +python tools/analyze_csv.py ~/Desktop/measure.csv --true 150 + +# Ablation — 5가지 알고리즘 config 비교 (어느 fix 가 임팩트 큰지) +python tools/analyze_csv.py measure.csv --true 150 --ablation + +# 시각화 plot 저장 +python tools/analyze_csv.py measure.csv --true 195 --plot ./out + +# 특정 cycle 디테일 (per-channel walls + signals) +python tools/analyze_csv.py measure.csv --true 150 --cycle 42 +``` + +### 출력 예시 (compare mode) + +``` +=== BV comparison === + APP (Kotlin): mean= 148.1± 4.2 trim10%= 148.5 bias= -1.9 ( -1.3%) CV= 2.9% lr mean=1.154 + Python : mean= 92.7±13.4 trim10%= 91.1 bias= -57.3 (-38.2%) CV=14.4% lr mean=0.680 + +=== App ↔ Python diff === + mean= +55.03 std=15.83 range=[-45.3, +74.7] + |diff| < 5 mL: 5/551 (0.9%) +``` + +### Ablation Config 종류 + +| 코드 | DPS | lr_ratio | 설명 | +|---|---|---|---| +| A | 1.936 | =1.0 강제 | OLD Kotlin equivalent (sim) | +| B | 1.981 | =1.0 강제 | DPS fix 단독 | +| C | 1.981 | Python 알고리즘 | **현재 b733d4f 결과** | +| D | 1.981 | Python + floor 1.0 | hybrid | +| E | 1.936 | Python 알고리즘 | lr 단독 영향 | + +### CSV 포맷 (AdcCsvLogger 기준) + +``` +scan_id, timestamp, ..., volume_ml, lr_ratio, ..., channel, s0..s99 +``` +한 scan = 6 rows (CH0..CH5), 100 samples / row. + +### 활용 예 + +1. **임상 BV 검증**: catheter 직후 측정 → `--true ` 로 정확도 비교 +2. **알고리즘 변경 검증**: 새 fix 후 같은 CSV 로 `--ablation` 돌려 회귀 여부 확인 +3. **Device-to-device variance**: 두 device 의 같은 phantom 측정 → CV 비교 +4. **lr_ratio 패턴**: human cohort 의 `--plot` 으로 lr 분포 시각화 diff --git a/tools/analyze_csv.py b/tools/analyze_csv.py new file mode 100644 index 0000000..be250bf --- /dev/null +++ b/tools/analyze_csv.py @@ -0,0 +1,390 @@ +""" +VesiScan CSV Analyzer — App 측정 결과 (raw ADC + computed BV) 를 Python pipeline +(appshare `for_app_share`) 으로 재계산하고 비교. + +Usage: + python tools/analyze_csv.py --true [options] + +Options: + --true VOL Known phantom/catheter volume (mL). Required. + --ablation Run 5 algorithm-config comparison (DPS, lr floor 등) + --plot DIR Save matplotlib plots to DIR (BV timeseries, lr dist, per-ch) + --cycle SID Inspect single scan_id detail (walls, signals) + --appshare PATH Path to appshare repo. Default $APPSHARE_DIR or + C:/Projects/appshare/piezo-phantom-test + +Examples: + # Basic compare on phantom 150 mL CSV + python tools/analyze_csv.py ~/Desktop/measure.csv --true 150 + + # With ablation (which fix breaks what) + python tools/analyze_csv.py ~/Desktop/measure.csv --true 150 --ablation + + # With plots + python tools/analyze_csv.py ~/Desktop/measure.csv --true 195 --plot ./out + + # Inspect specific cycle + python tools/analyze_csv.py ~/Desktop/measure.csv --true 150 --cycle 42 + +CSV format (from AdcCsvLogger): + scan_id, timestamp, ..., volume_ml, lr_ratio, ..., channel, s0..s99 + One row per channel (CH0..CH5), 100 samples per row. +""" +from __future__ import annotations + +import argparse +import csv +import os +import sys +from collections import defaultdict +from pathlib import Path +from statistics import mean, median, stdev + + +def _find_appshare(arg_path: str | None) -> Path: + candidates = [] + if arg_path: + candidates.append(Path(arg_path)) + env = os.environ.get("APPSHARE_DIR") + if env: + candidates.append(Path(env)) + candidates.extend([ + Path(r"C:/Projects/appshare/piezo-phantom-test"), + Path.home() / "Projects/appshare/piezo-phantom-test", + Path.home() / "appshare/piezo-phantom-test", + ]) + for c in candidates: + if (c / "for_app_share" / "__init__.py").exists(): + return c + sys.exit(f"ERROR: appshare repo not found. Tried: {candidates}\n" + f" Set --appshare PATH or APPSHARE_DIR env var.") + + +def load_scans(csv_path: Path) -> dict: + """Group CSV rows by scan_id → {sid: {meta, channels: {ch: [samples]}}}""" + scans: dict = defaultdict(lambda: {"meta": None, "channels": {}}) + with open(csv_path, encoding="utf-8") as f: + for row in csv.DictReader(f): + sid = int(row["scan_id"]) + ch = int(row["channel"].replace("CH", "")) + samples = [] + for i in range(100): + v = row.get(f"s{i}", "").strip() + if v == "": + break + try: + samples.append(int(v)) + except ValueError: + break + scans[sid]["channels"][ch] = samples + if scans[sid]["meta"] is None: + scans[sid]["meta"] = { + "timestamp": row.get("timestamp"), + "app_vol": float(row["volume_ml"]) if row.get("volume_ml") else None, + "app_lr": float(row["lr_ratio"]) if row.get("lr_ratio") else None, + "fw": row.get("firmware_version", ""), + "dps": float(row["dps"]) if row.get("dps") else None, + } + return scans + + +def trimmed_mean(vals, trim_frac: float = 0.1): + if not vals: + return None + if len(vals) < 5: + return mean(vals) + s = sorted(vals) + k = max(1, int(len(s) * trim_frac)) + return mean(s[k:len(s) - k]) + + +# ── Mode 1: basic compare ──────────────────────────────────────────────────── + +def mode_compare(scans, true_vol, py): + """App-computed BV vs Python-recomputed BV per scan.""" + import numpy as np + valid = [] + for sid, s in sorted(scans.items()): + if len(s["channels"]) != 6: + continue + if any(len(s["channels"][c]) < 100 for c in range(6)): + continue + sigs = np.stack([np.asarray(s["channels"][ch], dtype=float) for ch in range(6)]) + walls = py.detect_walls_multichannel(sigs) + walls_py = [None if w is None or w[0] is None or w[1] is None else tuple(w) + for w in walls] + bv = py.estimate_bladder_volume_6ch(walls_py) + valid.append({ + "sid": sid, + "app_vol": s["meta"]["app_vol"], + "app_lr": s["meta"]["app_lr"], + "py_vol": bv.volume_ml if bv else None, + "py_lr": bv.lr_ratio if bv else None, + "center": sum(1 for i in (0, 1, 2, 3) if walls_py[i] is not None), + "lateral": sum(1 for i in (4, 5) if walls_py[i] is not None), + }) + + app_vols = [r["app_vol"] for r in valid if r["app_vol"] is not None] + py_vols = [r["py_vol"] for r in valid if r["py_vol"] is not None] + print(f"Valid scans: {len(valid)} (true volume = {true_vol} mL)\n") + + def stats(label, vals, lrs=None): + if not vals: + print(f" {label}: no data"); return + m = mean(vals); sd = stdev(vals) if len(vals) > 1 else 0 + b = m - true_vol + cv = sd / m * 100 if m else 0 + tm = trimmed_mean(vals) + lr_str = "" + if lrs: + lr_clean = [x for x in lrs if x is not None] + if lr_clean: + lr_str = f" lr mean={mean(lr_clean):.3f}" + print(f" {label:>12}: mean={m:>6.1f}±{sd:>4.1f} trim10%={tm:>6.1f} " + f"bias={b:>+6.1f} ({b/true_vol*100:>+5.1f}%) CV={cv:>4.1f}%{lr_str}") + + print("=== BV comparison ===") + stats("APP (Kotlin)", app_vols, [r["app_lr"] for r in valid]) + stats("Python", py_vols, [r["py_lr"] for r in valid]) + + if app_vols and py_vols: + diffs = [a - p for a, p in zip(app_vols, py_vols)] + print(f"\n=== App ↔ Python diff ===") + print(f" mean={mean(diffs):>+6.2f} std={stdev(diffs):>5.2f} " + f"range=[{min(diffs):>+6.1f}, {max(diffs):>+6.1f}]") + print(f" |diff| < 5 mL: {sum(1 for d in diffs if abs(d) < 5)}/{len(diffs)} " + f"({sum(1 for d in diffs if abs(d) < 5) / len(diffs) * 100:.1f}%)") + print(f" |diff| < 1 mL: {sum(1 for d in diffs if abs(d) < 1)}/{len(diffs)}") + + print(f"\n=== Method D detection ===") + print(f" 4/4 center: {sum(1 for r in valid if r['center'] == 4) / len(valid) * 100:.1f}%") + print(f" >=3/4 center: {sum(1 for r in valid if r['center'] >= 3) / len(valid) * 100:.1f}%") + print(f" 2/2 lateral: {sum(1 for r in valid if r['lateral'] == 2) / len(valid) * 100:.1f}%") + + return valid + + +# ── Mode 2: ablation ───────────────────────────────────────────────────────── + +CONFIGS = [ + ("A. OLD eq (DPS=1.936, lr=1.0)", dict(distance_per_sample=1.936, lr_ratio_override=1.0)), + ("B. DPS fix only", dict(distance_per_sample=1.981, lr_ratio_override=1.0)), + ("C. NEW full (= b733d4f)", dict(distance_per_sample=1.981, lr_ratio_override=None)), + ("D. NEW + lr floor 1.0", "hybrid_floor"), + ("E. OLD DPS + new lr", dict(distance_per_sample=1.936, lr_ratio_override=None)), +] + + +def mode_ablation(scans, true_vol, py): + """Compare 5 algorithm configurations to isolate which fix matters.""" + import numpy as np + + def bv_with(walls, cfg): + if cfg == "hybrid_floor": + r = py.estimate_bladder_volume_6ch(walls, distance_per_sample=1.981) + if r is None: + return None + return py.estimate_bladder_volume_6ch( + walls, distance_per_sample=1.981, + lr_ratio_override=max(r.lr_ratio, 1.0)) + return py.estimate_bladder_volume_6ch(walls, **cfg) + + results = {name: ([], []) for name, _ in CONFIGS} # (vols, lrs) + app_vols = [] + valid_n = 0 + for sid, s in sorted(scans.items()): + if len(s["channels"]) != 6: + continue + if any(len(s["channels"][c]) < 100 for c in range(6)): + continue + sigs = np.stack([np.asarray(s["channels"][ch], dtype=float) for ch in range(6)]) + walls = py.detect_walls_multichannel(sigs) + walls_py = [None if w is None or w[0] is None or w[1] is None else tuple(w) + for w in walls] + center = sum(1 for i in (0, 1, 2, 3) if walls_py[i] is not None) + if center < 4: + continue + valid_n += 1 + for name, cfg in CONFIGS: + r = bv_with(walls_py, cfg) + if r is not None: + results[name][0].append(r.volume_ml) + results[name][1].append(r.lr_ratio) + if s["meta"]["app_vol"] is not None: + app_vols.append(s["meta"]["app_vol"]) + + print(f"Valid scans: {valid_n} (true = {true_vol} mL)\n") + print(f"{'config':<32} {'mean':>7} {'std':>6} {'bias':>7} {'%':>6} {'CV%':>5} {'lr':>6}") + print("-" * 90) + if app_vols: + m = mean(app_vols); sd = stdev(app_vols) if len(app_vols) > 1 else 0 + b = m - true_vol; cv = sd / m * 100 if m else 0 + print(f"{'APP (Kotlin, OLD APK)':<32} {m:>7.1f} {sd:>6.1f} {b:>+7.1f} " + f"{b/true_vol*100:>+5.1f}% {cv:>4.1f}% -") + for name, _ in CONFIGS: + vs, lrs = results[name] + if not vs: + continue + m = mean(vs); sd = stdev(vs) if len(vs) > 1 else 0 + b = m - true_vol; cv = sd / m * 100 if m else 0 + lr_clean = [x for x in lrs if x is not None] + lr_str = f"{mean(lr_clean):.3f}" if lr_clean else " -" + print(f"{name:<32} {m:>7.1f} {sd:>6.1f} {b:>+7.1f} " + f"{b/true_vol*100:>+5.1f}% {cv:>4.1f}% {lr_str:>6}") + + +# ── Mode 3: plot ───────────────────────────────────────────────────────────── + +def mode_plot(valid, true_vol, out_dir: Path): + try: + import matplotlib.pyplot as plt + except ImportError: + print("matplotlib not installed — skipping plots. pip install matplotlib") + return + out_dir.mkdir(parents=True, exist_ok=True) + + sids = [r["sid"] for r in valid] + app_vols = [r["app_vol"] if r["app_vol"] is not None else float("nan") for r in valid] + py_vols = [r["py_vol"] if r["py_vol"] is not None else float("nan") for r in valid] + + # 1. BV timeseries + fig, ax = plt.subplots(figsize=(12, 5)) + ax.plot(sids, app_vols, "-", lw=1, alpha=0.6, label="APP") + ax.plot(sids, py_vols, "-", lw=1, alpha=0.6, label="Python") + ax.axhline(true_vol, ls="--", color="g", label=f"true = {true_vol} mL") + ax.set_xlabel("scan_id"); ax.set_ylabel("BV (mL)") + ax.set_title(f"BV time series (n={len(sids)})") + ax.legend(); ax.grid(alpha=0.3) + fig.tight_layout() + fig.savefig(out_dir / "01_bv_timeseries.png", dpi=120) + plt.close(fig) + + # 2. App vs Python scatter + app_clean = [a for a, p in zip(app_vols, py_vols) + if not (a != a or p != p)] + py_clean = [p for a, p in zip(app_vols, py_vols) + if not (a != a or p != p)] + fig, ax = plt.subplots(figsize=(7, 7)) + ax.scatter(app_clean, py_clean, alpha=0.4, s=10) + lo, hi = min(app_clean + py_clean), max(app_clean + py_clean) + ax.plot([lo, hi], [lo, hi], "r--", lw=1, label="y=x") + ax.axhline(true_vol, ls=":", color="g", alpha=0.5) + ax.axvline(true_vol, ls=":", color="g", alpha=0.5, label=f"true {true_vol} mL") + ax.set_xlabel("APP BV (mL)"); ax.set_ylabel("Python BV (mL)") + ax.set_title("APP vs Python — per-scan agreement") + ax.legend(); ax.grid(alpha=0.3); ax.set_aspect("equal") + fig.tight_layout() + fig.savefig(out_dir / "02_app_vs_python_scatter.png", dpi=120) + plt.close(fig) + + # 3. lr_ratio distribution + py_lrs = [r["py_lr"] for r in valid if r["py_lr"] is not None] + app_lrs = [r["app_lr"] for r in valid if r["app_lr"] is not None] + fig, ax = plt.subplots(figsize=(10, 4)) + bins = 50 + if app_lrs: + ax.hist(app_lrs, bins=bins, alpha=0.5, label=f"APP (n={len(app_lrs)})") + if py_lrs: + ax.hist(py_lrs, bins=bins, alpha=0.5, label=f"Python (n={len(py_lrs)})") + ax.axvline(1.0, ls="--", color="k", alpha=0.5, label="lr = 1.0") + ax.set_xlabel("lr_ratio"); ax.set_ylabel("count") + ax.set_title("lr_ratio distribution") + ax.legend(); ax.grid(alpha=0.3) + fig.tight_layout() + fig.savefig(out_dir / "03_lr_ratio_hist.png", dpi=120) + plt.close(fig) + + print(f"Plots saved: {out_dir.resolve()}") + + +# ── Mode 4: single cycle inspect ───────────────────────────────────────────── + +def mode_cycle(scans, sid, py): + if sid not in scans: + print(f"scan_id={sid} not found. Range: {min(scans)}..{max(scans)}") + return + import numpy as np + s = scans[sid] + print(f"=== Scan {sid} ({s['meta']['timestamp']}) ===") + print(f"App BV: {s['meta']['app_vol']} mL lr={s['meta']['app_lr']}\n") + + sigs = np.stack([np.asarray(s["channels"][ch], dtype=float) for ch in range(6)]) + walls = py.detect_walls_multichannel(sigs) + print(f"{'CH':>3} {'peak':>5} {'idx':>4} {'ant':>4} {'post':>5} {'urine_len':>10}") + for ch in range(6): + sig = sigs[ch] + peak = int(sig.max()) if len(sig) else 0 + pidx = int(sig.argmax()) if len(sig) else 0 + w = walls[ch] + if w is None or w[0] is None: + print(f"{ch:>3} {peak:>5} {pidx:>4} - - -") + else: + ul = w[1] - w[0] + print(f"{ch:>3} {peak:>5} {pidx:>4} {w[0]:>4} {w[1]:>5} {ul:>10}") + + walls_py = [None if w is None or w[0] is None or w[1] is None else tuple(w) + for w in walls] + bv = py.estimate_bladder_volume_6ch(walls_py) + if bv: + print(f"\nPython BV: {bv.volume_ml:.1f} mL lr={bv.lr_ratio:.3f}") + else: + print("\nPython BV: failed (< 4 center walls or fit failed)") + + +def main(): + p = argparse.ArgumentParser( + description="VesiScan CSV analyzer", + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog=__doc__) + p.add_argument("csv", type=Path, help="App-exported CSV path") + p.add_argument("--true", dest="true_vol", type=float, required=True, + help="Known true volume (mL)") + p.add_argument("--ablation", action="store_true", + help="Run 5 algorithm-config comparison") + p.add_argument("--plot", type=Path, metavar="DIR", + help="Save plots to DIR") + p.add_argument("--cycle", type=int, metavar="SID", + help="Inspect single scan_id detail") + p.add_argument("--appshare", type=str, default=None, + help="Path to appshare repo (default: auto-detect)") + args = p.parse_args() + + if not args.csv.exists(): + sys.exit(f"CSV not found: {args.csv}") + + appshare_dir = _find_appshare(args.appshare) + sys.path.insert(0, str(appshare_dir)) + + # Lazy imports so --help works without numpy etc. + import numpy as np # noqa + import for_app_share as _fas + import vesiscan_test.library.method_d as _md + + class _PyAPI: + pass + _PyAPI.estimate_bladder_volume_6ch = staticmethod(_fas.estimate_bladder_volume_6ch) + _PyAPI.detect_walls_multichannel = staticmethod(_md.detect_walls_multichannel) + + print(f"Loading {args.csv.name}...") + scans = load_scans(args.csv) + print(f" {len(scans)} scans\n") + + if args.cycle is not None: + mode_cycle(scans, args.cycle, _PyAPI) + return + + if args.ablation: + print("─" * 60); print("ABLATION MODE"); print("─" * 60) + mode_ablation(scans, args.true_vol, _PyAPI) + print() + + print("─" * 60); print("COMPARE MODE"); print("─" * 60) + valid = mode_compare(scans, args.true_vol, _PyAPI) + + if args.plot: + print() + mode_plot(valid, args.true_vol, args.plot) + + +if __name__ == "__main__": + main()