Files
VesiscanClinicalAndroid/tools/analyze_csv.py
T
dw.jang 252692417a tools: analyze_csv.py — App CSV → Python pipeline 비교/검증 CLI
App 측정 시 자동 저장되는 CSV (Downloads/VesiScan_ADC/*.csv) 를
appshare for_app_share / vesiscan_test.library.method_d 로 재계산해
다음을 비교/시각화하는 영구 분석 툴.

4 modes:
  1. Basic compare: App BV vs Python BV (mean/std/bias/CV)
  2. Ablation: 5 algorithm config (DPS, lr floor 등) 개별 영향
  3. Plot: BV timeseries / app↔py scatter / lr_ratio histogram
  4. Cycle inspect: 특정 scan_id 의 raw → walls → BV detail

설계 의도:
  - 알고리즘 변경 후 회귀 자동 검증
  - 임상 측정 정확도 (catheter ground truth) 비교
  - device-to-device variance 분석
  - lr_ratio 분포로 phantom vs human anatomy 패턴 차이 확인

appshare 경로 자동 탐색 + APPSHARE_DIR env var 지원.

검증: phantom 150 mL CSV (2547 scans) 로 smoke test 통과.
  APP mean=147.6 (-1.6%), Python NEW mean=93.0 (-38%, lr=0.67)
  → ablation 으로 lr_ratio 가 차이의 주범 확인.

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
2026-06-30 17:49:28 +09:00

391 lines
16 KiB
Python

"""
VesiScan CSV Analyzer — App 측정 결과 (raw ADC + computed BV) 를 Python pipeline
(appshare `for_app_share`) 으로 재계산하고 비교.
Usage:
python tools/analyze_csv.py <csv_path> --true <known_volume_ml> [options]
Options:
--true VOL Known phantom/catheter volume (mL). Required.
--ablation Run 5 algorithm-config comparison (DPS, lr floor 등)
--plot DIR Save matplotlib plots to DIR (BV timeseries, lr dist, per-ch)
--cycle SID Inspect single scan_id detail (walls, signals)
--appshare PATH Path to appshare repo. Default $APPSHARE_DIR or
C:/Projects/appshare/piezo-phantom-test
Examples:
# Basic compare on phantom 150 mL CSV
python tools/analyze_csv.py ~/Desktop/measure.csv --true 150
# With ablation (which fix breaks what)
python tools/analyze_csv.py ~/Desktop/measure.csv --true 150 --ablation
# With plots
python tools/analyze_csv.py ~/Desktop/measure.csv --true 195 --plot ./out
# Inspect specific cycle
python tools/analyze_csv.py ~/Desktop/measure.csv --true 150 --cycle 42
CSV format (from AdcCsvLogger):
scan_id, timestamp, ..., volume_ml, lr_ratio, ..., channel, s0..s99
One row per channel (CH0..CH5), 100 samples per row.
"""
from __future__ import annotations
import argparse
import csv
import os
import sys
from collections import defaultdict
from pathlib import Path
from statistics import mean, median, stdev
def _find_appshare(arg_path: str | None) -> Path:
candidates = []
if arg_path:
candidates.append(Path(arg_path))
env = os.environ.get("APPSHARE_DIR")
if env:
candidates.append(Path(env))
candidates.extend([
Path(r"C:/Projects/appshare/piezo-phantom-test"),
Path.home() / "Projects/appshare/piezo-phantom-test",
Path.home() / "appshare/piezo-phantom-test",
])
for c in candidates:
if (c / "for_app_share" / "__init__.py").exists():
return c
sys.exit(f"ERROR: appshare repo not found. Tried: {candidates}\n"
f" Set --appshare PATH or APPSHARE_DIR env var.")
def load_scans(csv_path: Path) -> dict:
"""Group CSV rows by scan_id → {sid: {meta, channels: {ch: [samples]}}}"""
scans: dict = defaultdict(lambda: {"meta": None, "channels": {}})
with open(csv_path, encoding="utf-8") as f:
for row in csv.DictReader(f):
sid = int(row["scan_id"])
ch = int(row["channel"].replace("CH", ""))
samples = []
for i in range(100):
v = row.get(f"s{i}", "").strip()
if v == "":
break
try:
samples.append(int(v))
except ValueError:
break
scans[sid]["channels"][ch] = samples
if scans[sid]["meta"] is None:
scans[sid]["meta"] = {
"timestamp": row.get("timestamp"),
"app_vol": float(row["volume_ml"]) if row.get("volume_ml") else None,
"app_lr": float(row["lr_ratio"]) if row.get("lr_ratio") else None,
"fw": row.get("firmware_version", ""),
"dps": float(row["dps"]) if row.get("dps") else None,
}
return scans
def trimmed_mean(vals, trim_frac: float = 0.1):
if not vals:
return None
if len(vals) < 5:
return mean(vals)
s = sorted(vals)
k = max(1, int(len(s) * trim_frac))
return mean(s[k:len(s) - k])
# ── Mode 1: basic compare ────────────────────────────────────────────────────
def mode_compare(scans, true_vol, py):
"""App-computed BV vs Python-recomputed BV per scan."""
import numpy as np
valid = []
for sid, s in sorted(scans.items()):
if len(s["channels"]) != 6:
continue
if any(len(s["channels"][c]) < 100 for c in range(6)):
continue
sigs = np.stack([np.asarray(s["channels"][ch], dtype=float) for ch in range(6)])
walls = py.detect_walls_multichannel(sigs)
walls_py = [None if w is None or w[0] is None or w[1] is None else tuple(w)
for w in walls]
bv = py.estimate_bladder_volume_6ch(walls_py)
valid.append({
"sid": sid,
"app_vol": s["meta"]["app_vol"],
"app_lr": s["meta"]["app_lr"],
"py_vol": bv.volume_ml if bv else None,
"py_lr": bv.lr_ratio if bv else None,
"center": sum(1 for i in (0, 1, 2, 3) if walls_py[i] is not None),
"lateral": sum(1 for i in (4, 5) if walls_py[i] is not None),
})
app_vols = [r["app_vol"] for r in valid if r["app_vol"] is not None]
py_vols = [r["py_vol"] for r in valid if r["py_vol"] is not None]
print(f"Valid scans: {len(valid)} (true volume = {true_vol} mL)\n")
def stats(label, vals, lrs=None):
if not vals:
print(f" {label}: no data"); return
m = mean(vals); sd = stdev(vals) if len(vals) > 1 else 0
b = m - true_vol
cv = sd / m * 100 if m else 0
tm = trimmed_mean(vals)
lr_str = ""
if lrs:
lr_clean = [x for x in lrs if x is not None]
if lr_clean:
lr_str = f" lr mean={mean(lr_clean):.3f}"
print(f" {label:>12}: mean={m:>6.1f}±{sd:>4.1f} trim10%={tm:>6.1f} "
f"bias={b:>+6.1f} ({b/true_vol*100:>+5.1f}%) CV={cv:>4.1f}%{lr_str}")
print("=== BV comparison ===")
stats("APP (Kotlin)", app_vols, [r["app_lr"] for r in valid])
stats("Python", py_vols, [r["py_lr"] for r in valid])
if app_vols and py_vols:
diffs = [a - p for a, p in zip(app_vols, py_vols)]
print(f"\n=== App ↔ Python diff ===")
print(f" mean={mean(diffs):>+6.2f} std={stdev(diffs):>5.2f} "
f"range=[{min(diffs):>+6.1f}, {max(diffs):>+6.1f}]")
print(f" |diff| < 5 mL: {sum(1 for d in diffs if abs(d) < 5)}/{len(diffs)} "
f"({sum(1 for d in diffs if abs(d) < 5) / len(diffs) * 100:.1f}%)")
print(f" |diff| < 1 mL: {sum(1 for d in diffs if abs(d) < 1)}/{len(diffs)}")
print(f"\n=== Method D detection ===")
print(f" 4/4 center: {sum(1 for r in valid if r['center'] == 4) / len(valid) * 100:.1f}%")
print(f" >=3/4 center: {sum(1 for r in valid if r['center'] >= 3) / len(valid) * 100:.1f}%")
print(f" 2/2 lateral: {sum(1 for r in valid if r['lateral'] == 2) / len(valid) * 100:.1f}%")
return valid
# ── Mode 2: ablation ─────────────────────────────────────────────────────────
CONFIGS = [
("A. OLD eq (DPS=1.936, lr=1.0)", dict(distance_per_sample=1.936, lr_ratio_override=1.0)),
("B. DPS fix only", dict(distance_per_sample=1.981, lr_ratio_override=1.0)),
("C. NEW full (= b733d4f)", dict(distance_per_sample=1.981, lr_ratio_override=None)),
("D. NEW + lr floor 1.0", "hybrid_floor"),
("E. OLD DPS + new lr", dict(distance_per_sample=1.936, lr_ratio_override=None)),
]
def mode_ablation(scans, true_vol, py):
"""Compare 5 algorithm configurations to isolate which fix matters."""
import numpy as np
def bv_with(walls, cfg):
if cfg == "hybrid_floor":
r = py.estimate_bladder_volume_6ch(walls, distance_per_sample=1.981)
if r is None:
return None
return py.estimate_bladder_volume_6ch(
walls, distance_per_sample=1.981,
lr_ratio_override=max(r.lr_ratio, 1.0))
return py.estimate_bladder_volume_6ch(walls, **cfg)
results = {name: ([], []) for name, _ in CONFIGS} # (vols, lrs)
app_vols = []
valid_n = 0
for sid, s in sorted(scans.items()):
if len(s["channels"]) != 6:
continue
if any(len(s["channels"][c]) < 100 for c in range(6)):
continue
sigs = np.stack([np.asarray(s["channels"][ch], dtype=float) for ch in range(6)])
walls = py.detect_walls_multichannel(sigs)
walls_py = [None if w is None or w[0] is None or w[1] is None else tuple(w)
for w in walls]
center = sum(1 for i in (0, 1, 2, 3) if walls_py[i] is not None)
if center < 4:
continue
valid_n += 1
for name, cfg in CONFIGS:
r = bv_with(walls_py, cfg)
if r is not None:
results[name][0].append(r.volume_ml)
results[name][1].append(r.lr_ratio)
if s["meta"]["app_vol"] is not None:
app_vols.append(s["meta"]["app_vol"])
print(f"Valid scans: {valid_n} (true = {true_vol} mL)\n")
print(f"{'config':<32} {'mean':>7} {'std':>6} {'bias':>7} {'%':>6} {'CV%':>5} {'lr':>6}")
print("-" * 90)
if app_vols:
m = mean(app_vols); sd = stdev(app_vols) if len(app_vols) > 1 else 0
b = m - true_vol; cv = sd / m * 100 if m else 0
print(f"{'APP (Kotlin, OLD APK)':<32} {m:>7.1f} {sd:>6.1f} {b:>+7.1f} "
f"{b/true_vol*100:>+5.1f}% {cv:>4.1f}% -")
for name, _ in CONFIGS:
vs, lrs = results[name]
if not vs:
continue
m = mean(vs); sd = stdev(vs) if len(vs) > 1 else 0
b = m - true_vol; cv = sd / m * 100 if m else 0
lr_clean = [x for x in lrs if x is not None]
lr_str = f"{mean(lr_clean):.3f}" if lr_clean else " -"
print(f"{name:<32} {m:>7.1f} {sd:>6.1f} {b:>+7.1f} "
f"{b/true_vol*100:>+5.1f}% {cv:>4.1f}% {lr_str:>6}")
# ── Mode 3: plot ─────────────────────────────────────────────────────────────
def mode_plot(valid, true_vol, out_dir: Path):
try:
import matplotlib.pyplot as plt
except ImportError:
print("matplotlib not installed — skipping plots. pip install matplotlib")
return
out_dir.mkdir(parents=True, exist_ok=True)
sids = [r["sid"] for r in valid]
app_vols = [r["app_vol"] if r["app_vol"] is not None else float("nan") for r in valid]
py_vols = [r["py_vol"] if r["py_vol"] is not None else float("nan") for r in valid]
# 1. BV timeseries
fig, ax = plt.subplots(figsize=(12, 5))
ax.plot(sids, app_vols, "-", lw=1, alpha=0.6, label="APP")
ax.plot(sids, py_vols, "-", lw=1, alpha=0.6, label="Python")
ax.axhline(true_vol, ls="--", color="g", label=f"true = {true_vol} mL")
ax.set_xlabel("scan_id"); ax.set_ylabel("BV (mL)")
ax.set_title(f"BV time series (n={len(sids)})")
ax.legend(); ax.grid(alpha=0.3)
fig.tight_layout()
fig.savefig(out_dir / "01_bv_timeseries.png", dpi=120)
plt.close(fig)
# 2. App vs Python scatter
app_clean = [a for a, p in zip(app_vols, py_vols)
if not (a != a or p != p)]
py_clean = [p for a, p in zip(app_vols, py_vols)
if not (a != a or p != p)]
fig, ax = plt.subplots(figsize=(7, 7))
ax.scatter(app_clean, py_clean, alpha=0.4, s=10)
lo, hi = min(app_clean + py_clean), max(app_clean + py_clean)
ax.plot([lo, hi], [lo, hi], "r--", lw=1, label="y=x")
ax.axhline(true_vol, ls=":", color="g", alpha=0.5)
ax.axvline(true_vol, ls=":", color="g", alpha=0.5, label=f"true {true_vol} mL")
ax.set_xlabel("APP BV (mL)"); ax.set_ylabel("Python BV (mL)")
ax.set_title("APP vs Python — per-scan agreement")
ax.legend(); ax.grid(alpha=0.3); ax.set_aspect("equal")
fig.tight_layout()
fig.savefig(out_dir / "02_app_vs_python_scatter.png", dpi=120)
plt.close(fig)
# 3. lr_ratio distribution
py_lrs = [r["py_lr"] for r in valid if r["py_lr"] is not None]
app_lrs = [r["app_lr"] for r in valid if r["app_lr"] is not None]
fig, ax = plt.subplots(figsize=(10, 4))
bins = 50
if app_lrs:
ax.hist(app_lrs, bins=bins, alpha=0.5, label=f"APP (n={len(app_lrs)})")
if py_lrs:
ax.hist(py_lrs, bins=bins, alpha=0.5, label=f"Python (n={len(py_lrs)})")
ax.axvline(1.0, ls="--", color="k", alpha=0.5, label="lr = 1.0")
ax.set_xlabel("lr_ratio"); ax.set_ylabel("count")
ax.set_title("lr_ratio distribution")
ax.legend(); ax.grid(alpha=0.3)
fig.tight_layout()
fig.savefig(out_dir / "03_lr_ratio_hist.png", dpi=120)
plt.close(fig)
print(f"Plots saved: {out_dir.resolve()}")
# ── Mode 4: single cycle inspect ─────────────────────────────────────────────
def mode_cycle(scans, sid, py):
if sid not in scans:
print(f"scan_id={sid} not found. Range: {min(scans)}..{max(scans)}")
return
import numpy as np
s = scans[sid]
print(f"=== Scan {sid} ({s['meta']['timestamp']}) ===")
print(f"App BV: {s['meta']['app_vol']} mL lr={s['meta']['app_lr']}\n")
sigs = np.stack([np.asarray(s["channels"][ch], dtype=float) for ch in range(6)])
walls = py.detect_walls_multichannel(sigs)
print(f"{'CH':>3} {'peak':>5} {'idx':>4} {'ant':>4} {'post':>5} {'urine_len':>10}")
for ch in range(6):
sig = sigs[ch]
peak = int(sig.max()) if len(sig) else 0
pidx = int(sig.argmax()) if len(sig) else 0
w = walls[ch]
if w is None or w[0] is None:
print(f"{ch:>3} {peak:>5} {pidx:>4} - - -")
else:
ul = w[1] - w[0]
print(f"{ch:>3} {peak:>5} {pidx:>4} {w[0]:>4} {w[1]:>5} {ul:>10}")
walls_py = [None if w is None or w[0] is None or w[1] is None else tuple(w)
for w in walls]
bv = py.estimate_bladder_volume_6ch(walls_py)
if bv:
print(f"\nPython BV: {bv.volume_ml:.1f} mL lr={bv.lr_ratio:.3f}")
else:
print("\nPython BV: failed (< 4 center walls or fit failed)")
def main():
p = argparse.ArgumentParser(
description="VesiScan CSV analyzer",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog=__doc__)
p.add_argument("csv", type=Path, help="App-exported CSV path")
p.add_argument("--true", dest="true_vol", type=float, required=True,
help="Known true volume (mL)")
p.add_argument("--ablation", action="store_true",
help="Run 5 algorithm-config comparison")
p.add_argument("--plot", type=Path, metavar="DIR",
help="Save plots to DIR")
p.add_argument("--cycle", type=int, metavar="SID",
help="Inspect single scan_id detail")
p.add_argument("--appshare", type=str, default=None,
help="Path to appshare repo (default: auto-detect)")
args = p.parse_args()
if not args.csv.exists():
sys.exit(f"CSV not found: {args.csv}")
appshare_dir = _find_appshare(args.appshare)
sys.path.insert(0, str(appshare_dir))
# Lazy imports so --help works without numpy etc.
import numpy as np # noqa
import for_app_share as _fas
import vesiscan_test.library.method_d as _md
class _PyAPI:
pass
_PyAPI.estimate_bladder_volume_6ch = staticmethod(_fas.estimate_bladder_volume_6ch)
_PyAPI.detect_walls_multichannel = staticmethod(_md.detect_walls_multichannel)
print(f"Loading {args.csv.name}...")
scans = load_scans(args.csv)
print(f" {len(scans)} scans\n")
if args.cycle is not None:
mode_cycle(scans, args.cycle, _PyAPI)
return
if args.ablation:
print("─" * 60); print("ABLATION MODE"); print("─" * 60)
mode_ablation(scans, args.true_vol, _PyAPI)
print()
print("─" * 60); print("COMPARE MODE"); print("─" * 60)
valid = mode_compare(scans, args.true_vol, _PyAPI)
if args.plot:
print()
mode_plot(valid, args.true_vol, args.plot)
if __name__ == "__main__":
main()