tools: analyze_csv.py — App CSV → Python pipeline 비교/검증 CLI
App 측정 시 자동 저장되는 CSV (Downloads/VesiScan_ADC/*.csv) 를 appshare for_app_share / vesiscan_test.library.method_d 로 재계산해 다음을 비교/시각화하는 영구 분석 툴. 4 modes: 1. Basic compare: App BV vs Python BV (mean/std/bias/CV) 2. Ablation: 5 algorithm config (DPS, lr floor 등) 개별 영향 3. Plot: BV timeseries / app↔py scatter / lr_ratio histogram 4. Cycle inspect: 특정 scan_id 의 raw → walls → BV detail 설계 의도: - 알고리즘 변경 후 회귀 자동 검증 - 임상 측정 정확도 (catheter ground truth) 비교 - device-to-device variance 분석 - lr_ratio 분포로 phantom vs human anatomy 패턴 차이 확인 appshare 경로 자동 탐색 + APPSHARE_DIR env var 지원. 검증: phantom 150 mL CSV (2547 scans) 로 smoke test 통과. APP mean=147.6 (-1.6%), Python NEW mean=93.0 (-38%, lr=0.67) → ablation 으로 lr_ratio 가 차이의 주범 확인. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,390 @@
|
||||
"""
|
||||
VesiScan CSV Analyzer — App 측정 결과 (raw ADC + computed BV) 를 Python pipeline
|
||||
(appshare `for_app_share`) 으로 재계산하고 비교.
|
||||
|
||||
Usage:
|
||||
python tools/analyze_csv.py <csv_path> --true <known_volume_ml> [options]
|
||||
|
||||
Options:
|
||||
--true VOL Known phantom/catheter volume (mL). Required.
|
||||
--ablation Run 5 algorithm-config comparison (DPS, lr floor 등)
|
||||
--plot DIR Save matplotlib plots to DIR (BV timeseries, lr dist, per-ch)
|
||||
--cycle SID Inspect single scan_id detail (walls, signals)
|
||||
--appshare PATH Path to appshare repo. Default $APPSHARE_DIR or
|
||||
C:/Projects/appshare/piezo-phantom-test
|
||||
|
||||
Examples:
|
||||
# Basic compare on phantom 150 mL CSV
|
||||
python tools/analyze_csv.py ~/Desktop/measure.csv --true 150
|
||||
|
||||
# With ablation (which fix breaks what)
|
||||
python tools/analyze_csv.py ~/Desktop/measure.csv --true 150 --ablation
|
||||
|
||||
# With plots
|
||||
python tools/analyze_csv.py ~/Desktop/measure.csv --true 195 --plot ./out
|
||||
|
||||
# Inspect specific cycle
|
||||
python tools/analyze_csv.py ~/Desktop/measure.csv --true 150 --cycle 42
|
||||
|
||||
CSV format (from AdcCsvLogger):
|
||||
scan_id, timestamp, ..., volume_ml, lr_ratio, ..., channel, s0..s99
|
||||
One row per channel (CH0..CH5), 100 samples per row.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
import os
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
from statistics import mean, median, stdev
|
||||
|
||||
|
||||
def _find_appshare(arg_path: str | None) -> Path:
|
||||
candidates = []
|
||||
if arg_path:
|
||||
candidates.append(Path(arg_path))
|
||||
env = os.environ.get("APPSHARE_DIR")
|
||||
if env:
|
||||
candidates.append(Path(env))
|
||||
candidates.extend([
|
||||
Path(r"C:/Projects/appshare/piezo-phantom-test"),
|
||||
Path.home() / "Projects/appshare/piezo-phantom-test",
|
||||
Path.home() / "appshare/piezo-phantom-test",
|
||||
])
|
||||
for c in candidates:
|
||||
if (c / "for_app_share" / "__init__.py").exists():
|
||||
return c
|
||||
sys.exit(f"ERROR: appshare repo not found. Tried: {candidates}\n"
|
||||
f" Set --appshare PATH or APPSHARE_DIR env var.")
|
||||
|
||||
|
||||
def load_scans(csv_path: Path) -> dict:
|
||||
"""Group CSV rows by scan_id → {sid: {meta, channels: {ch: [samples]}}}"""
|
||||
scans: dict = defaultdict(lambda: {"meta": None, "channels": {}})
|
||||
with open(csv_path, encoding="utf-8") as f:
|
||||
for row in csv.DictReader(f):
|
||||
sid = int(row["scan_id"])
|
||||
ch = int(row["channel"].replace("CH", ""))
|
||||
samples = []
|
||||
for i in range(100):
|
||||
v = row.get(f"s{i}", "").strip()
|
||||
if v == "":
|
||||
break
|
||||
try:
|
||||
samples.append(int(v))
|
||||
except ValueError:
|
||||
break
|
||||
scans[sid]["channels"][ch] = samples
|
||||
if scans[sid]["meta"] is None:
|
||||
scans[sid]["meta"] = {
|
||||
"timestamp": row.get("timestamp"),
|
||||
"app_vol": float(row["volume_ml"]) if row.get("volume_ml") else None,
|
||||
"app_lr": float(row["lr_ratio"]) if row.get("lr_ratio") else None,
|
||||
"fw": row.get("firmware_version", ""),
|
||||
"dps": float(row["dps"]) if row.get("dps") else None,
|
||||
}
|
||||
return scans
|
||||
|
||||
|
||||
def trimmed_mean(vals, trim_frac: float = 0.1):
|
||||
if not vals:
|
||||
return None
|
||||
if len(vals) < 5:
|
||||
return mean(vals)
|
||||
s = sorted(vals)
|
||||
k = max(1, int(len(s) * trim_frac))
|
||||
return mean(s[k:len(s) - k])
|
||||
|
||||
|
||||
# ── Mode 1: basic compare ────────────────────────────────────────────────────
|
||||
|
||||
def mode_compare(scans, true_vol, py):
|
||||
"""App-computed BV vs Python-recomputed BV per scan."""
|
||||
import numpy as np
|
||||
valid = []
|
||||
for sid, s in sorted(scans.items()):
|
||||
if len(s["channels"]) != 6:
|
||||
continue
|
||||
if any(len(s["channels"][c]) < 100 for c in range(6)):
|
||||
continue
|
||||
sigs = np.stack([np.asarray(s["channels"][ch], dtype=float) for ch in range(6)])
|
||||
walls = py.detect_walls_multichannel(sigs)
|
||||
walls_py = [None if w is None or w[0] is None or w[1] is None else tuple(w)
|
||||
for w in walls]
|
||||
bv = py.estimate_bladder_volume_6ch(walls_py)
|
||||
valid.append({
|
||||
"sid": sid,
|
||||
"app_vol": s["meta"]["app_vol"],
|
||||
"app_lr": s["meta"]["app_lr"],
|
||||
"py_vol": bv.volume_ml if bv else None,
|
||||
"py_lr": bv.lr_ratio if bv else None,
|
||||
"center": sum(1 for i in (0, 1, 2, 3) if walls_py[i] is not None),
|
||||
"lateral": sum(1 for i in (4, 5) if walls_py[i] is not None),
|
||||
})
|
||||
|
||||
app_vols = [r["app_vol"] for r in valid if r["app_vol"] is not None]
|
||||
py_vols = [r["py_vol"] for r in valid if r["py_vol"] is not None]
|
||||
print(f"Valid scans: {len(valid)} (true volume = {true_vol} mL)\n")
|
||||
|
||||
def stats(label, vals, lrs=None):
|
||||
if not vals:
|
||||
print(f" {label}: no data"); return
|
||||
m = mean(vals); sd = stdev(vals) if len(vals) > 1 else 0
|
||||
b = m - true_vol
|
||||
cv = sd / m * 100 if m else 0
|
||||
tm = trimmed_mean(vals)
|
||||
lr_str = ""
|
||||
if lrs:
|
||||
lr_clean = [x for x in lrs if x is not None]
|
||||
if lr_clean:
|
||||
lr_str = f" lr mean={mean(lr_clean):.3f}"
|
||||
print(f" {label:>12}: mean={m:>6.1f}±{sd:>4.1f} trim10%={tm:>6.1f} "
|
||||
f"bias={b:>+6.1f} ({b/true_vol*100:>+5.1f}%) CV={cv:>4.1f}%{lr_str}")
|
||||
|
||||
print("=== BV comparison ===")
|
||||
stats("APP (Kotlin)", app_vols, [r["app_lr"] for r in valid])
|
||||
stats("Python", py_vols, [r["py_lr"] for r in valid])
|
||||
|
||||
if app_vols and py_vols:
|
||||
diffs = [a - p for a, p in zip(app_vols, py_vols)]
|
||||
print(f"\n=== App ↔ Python diff ===")
|
||||
print(f" mean={mean(diffs):>+6.2f} std={stdev(diffs):>5.2f} "
|
||||
f"range=[{min(diffs):>+6.1f}, {max(diffs):>+6.1f}]")
|
||||
print(f" |diff| < 5 mL: {sum(1 for d in diffs if abs(d) < 5)}/{len(diffs)} "
|
||||
f"({sum(1 for d in diffs if abs(d) < 5) / len(diffs) * 100:.1f}%)")
|
||||
print(f" |diff| < 1 mL: {sum(1 for d in diffs if abs(d) < 1)}/{len(diffs)}")
|
||||
|
||||
print(f"\n=== Method D detection ===")
|
||||
print(f" 4/4 center: {sum(1 for r in valid if r['center'] == 4) / len(valid) * 100:.1f}%")
|
||||
print(f" >=3/4 center: {sum(1 for r in valid if r['center'] >= 3) / len(valid) * 100:.1f}%")
|
||||
print(f" 2/2 lateral: {sum(1 for r in valid if r['lateral'] == 2) / len(valid) * 100:.1f}%")
|
||||
|
||||
return valid
|
||||
|
||||
|
||||
# ── Mode 2: ablation ─────────────────────────────────────────────────────────
|
||||
|
||||
CONFIGS = [
|
||||
("A. OLD eq (DPS=1.936, lr=1.0)", dict(distance_per_sample=1.936, lr_ratio_override=1.0)),
|
||||
("B. DPS fix only", dict(distance_per_sample=1.981, lr_ratio_override=1.0)),
|
||||
("C. NEW full (= b733d4f)", dict(distance_per_sample=1.981, lr_ratio_override=None)),
|
||||
("D. NEW + lr floor 1.0", "hybrid_floor"),
|
||||
("E. OLD DPS + new lr", dict(distance_per_sample=1.936, lr_ratio_override=None)),
|
||||
]
|
||||
|
||||
|
||||
def mode_ablation(scans, true_vol, py):
|
||||
"""Compare 5 algorithm configurations to isolate which fix matters."""
|
||||
import numpy as np
|
||||
|
||||
def bv_with(walls, cfg):
|
||||
if cfg == "hybrid_floor":
|
||||
r = py.estimate_bladder_volume_6ch(walls, distance_per_sample=1.981)
|
||||
if r is None:
|
||||
return None
|
||||
return py.estimate_bladder_volume_6ch(
|
||||
walls, distance_per_sample=1.981,
|
||||
lr_ratio_override=max(r.lr_ratio, 1.0))
|
||||
return py.estimate_bladder_volume_6ch(walls, **cfg)
|
||||
|
||||
results = {name: ([], []) for name, _ in CONFIGS} # (vols, lrs)
|
||||
app_vols = []
|
||||
valid_n = 0
|
||||
for sid, s in sorted(scans.items()):
|
||||
if len(s["channels"]) != 6:
|
||||
continue
|
||||
if any(len(s["channels"][c]) < 100 for c in range(6)):
|
||||
continue
|
||||
sigs = np.stack([np.asarray(s["channels"][ch], dtype=float) for ch in range(6)])
|
||||
walls = py.detect_walls_multichannel(sigs)
|
||||
walls_py = [None if w is None or w[0] is None or w[1] is None else tuple(w)
|
||||
for w in walls]
|
||||
center = sum(1 for i in (0, 1, 2, 3) if walls_py[i] is not None)
|
||||
if center < 4:
|
||||
continue
|
||||
valid_n += 1
|
||||
for name, cfg in CONFIGS:
|
||||
r = bv_with(walls_py, cfg)
|
||||
if r is not None:
|
||||
results[name][0].append(r.volume_ml)
|
||||
results[name][1].append(r.lr_ratio)
|
||||
if s["meta"]["app_vol"] is not None:
|
||||
app_vols.append(s["meta"]["app_vol"])
|
||||
|
||||
print(f"Valid scans: {valid_n} (true = {true_vol} mL)\n")
|
||||
print(f"{'config':<32} {'mean':>7} {'std':>6} {'bias':>7} {'%':>6} {'CV%':>5} {'lr':>6}")
|
||||
print("-" * 90)
|
||||
if app_vols:
|
||||
m = mean(app_vols); sd = stdev(app_vols) if len(app_vols) > 1 else 0
|
||||
b = m - true_vol; cv = sd / m * 100 if m else 0
|
||||
print(f"{'APP (Kotlin, OLD APK)':<32} {m:>7.1f} {sd:>6.1f} {b:>+7.1f} "
|
||||
f"{b/true_vol*100:>+5.1f}% {cv:>4.1f}% -")
|
||||
for name, _ in CONFIGS:
|
||||
vs, lrs = results[name]
|
||||
if not vs:
|
||||
continue
|
||||
m = mean(vs); sd = stdev(vs) if len(vs) > 1 else 0
|
||||
b = m - true_vol; cv = sd / m * 100 if m else 0
|
||||
lr_clean = [x for x in lrs if x is not None]
|
||||
lr_str = f"{mean(lr_clean):.3f}" if lr_clean else " -"
|
||||
print(f"{name:<32} {m:>7.1f} {sd:>6.1f} {b:>+7.1f} "
|
||||
f"{b/true_vol*100:>+5.1f}% {cv:>4.1f}% {lr_str:>6}")
|
||||
|
||||
|
||||
# ── Mode 3: plot ─────────────────────────────────────────────────────────────
|
||||
|
||||
def mode_plot(valid, true_vol, out_dir: Path):
|
||||
try:
|
||||
import matplotlib.pyplot as plt
|
||||
except ImportError:
|
||||
print("matplotlib not installed — skipping plots. pip install matplotlib")
|
||||
return
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
sids = [r["sid"] for r in valid]
|
||||
app_vols = [r["app_vol"] if r["app_vol"] is not None else float("nan") for r in valid]
|
||||
py_vols = [r["py_vol"] if r["py_vol"] is not None else float("nan") for r in valid]
|
||||
|
||||
# 1. BV timeseries
|
||||
fig, ax = plt.subplots(figsize=(12, 5))
|
||||
ax.plot(sids, app_vols, "-", lw=1, alpha=0.6, label="APP")
|
||||
ax.plot(sids, py_vols, "-", lw=1, alpha=0.6, label="Python")
|
||||
ax.axhline(true_vol, ls="--", color="g", label=f"true = {true_vol} mL")
|
||||
ax.set_xlabel("scan_id"); ax.set_ylabel("BV (mL)")
|
||||
ax.set_title(f"BV time series (n={len(sids)})")
|
||||
ax.legend(); ax.grid(alpha=0.3)
|
||||
fig.tight_layout()
|
||||
fig.savefig(out_dir / "01_bv_timeseries.png", dpi=120)
|
||||
plt.close(fig)
|
||||
|
||||
# 2. App vs Python scatter
|
||||
app_clean = [a for a, p in zip(app_vols, py_vols)
|
||||
if not (a != a or p != p)]
|
||||
py_clean = [p for a, p in zip(app_vols, py_vols)
|
||||
if not (a != a or p != p)]
|
||||
fig, ax = plt.subplots(figsize=(7, 7))
|
||||
ax.scatter(app_clean, py_clean, alpha=0.4, s=10)
|
||||
lo, hi = min(app_clean + py_clean), max(app_clean + py_clean)
|
||||
ax.plot([lo, hi], [lo, hi], "r--", lw=1, label="y=x")
|
||||
ax.axhline(true_vol, ls=":", color="g", alpha=0.5)
|
||||
ax.axvline(true_vol, ls=":", color="g", alpha=0.5, label=f"true {true_vol} mL")
|
||||
ax.set_xlabel("APP BV (mL)"); ax.set_ylabel("Python BV (mL)")
|
||||
ax.set_title("APP vs Python — per-scan agreement")
|
||||
ax.legend(); ax.grid(alpha=0.3); ax.set_aspect("equal")
|
||||
fig.tight_layout()
|
||||
fig.savefig(out_dir / "02_app_vs_python_scatter.png", dpi=120)
|
||||
plt.close(fig)
|
||||
|
||||
# 3. lr_ratio distribution
|
||||
py_lrs = [r["py_lr"] for r in valid if r["py_lr"] is not None]
|
||||
app_lrs = [r["app_lr"] for r in valid if r["app_lr"] is not None]
|
||||
fig, ax = plt.subplots(figsize=(10, 4))
|
||||
bins = 50
|
||||
if app_lrs:
|
||||
ax.hist(app_lrs, bins=bins, alpha=0.5, label=f"APP (n={len(app_lrs)})")
|
||||
if py_lrs:
|
||||
ax.hist(py_lrs, bins=bins, alpha=0.5, label=f"Python (n={len(py_lrs)})")
|
||||
ax.axvline(1.0, ls="--", color="k", alpha=0.5, label="lr = 1.0")
|
||||
ax.set_xlabel("lr_ratio"); ax.set_ylabel("count")
|
||||
ax.set_title("lr_ratio distribution")
|
||||
ax.legend(); ax.grid(alpha=0.3)
|
||||
fig.tight_layout()
|
||||
fig.savefig(out_dir / "03_lr_ratio_hist.png", dpi=120)
|
||||
plt.close(fig)
|
||||
|
||||
print(f"Plots saved: {out_dir.resolve()}")
|
||||
|
||||
|
||||
# ── Mode 4: single cycle inspect ─────────────────────────────────────────────
|
||||
|
||||
def mode_cycle(scans, sid, py):
|
||||
if sid not in scans:
|
||||
print(f"scan_id={sid} not found. Range: {min(scans)}..{max(scans)}")
|
||||
return
|
||||
import numpy as np
|
||||
s = scans[sid]
|
||||
print(f"=== Scan {sid} ({s['meta']['timestamp']}) ===")
|
||||
print(f"App BV: {s['meta']['app_vol']} mL lr={s['meta']['app_lr']}\n")
|
||||
|
||||
sigs = np.stack([np.asarray(s["channels"][ch], dtype=float) for ch in range(6)])
|
||||
walls = py.detect_walls_multichannel(sigs)
|
||||
print(f"{'CH':>3} {'peak':>5} {'idx':>4} {'ant':>4} {'post':>5} {'urine_len':>10}")
|
||||
for ch in range(6):
|
||||
sig = sigs[ch]
|
||||
peak = int(sig.max()) if len(sig) else 0
|
||||
pidx = int(sig.argmax()) if len(sig) else 0
|
||||
w = walls[ch]
|
||||
if w is None or w[0] is None:
|
||||
print(f"{ch:>3} {peak:>5} {pidx:>4} - - -")
|
||||
else:
|
||||
ul = w[1] - w[0]
|
||||
print(f"{ch:>3} {peak:>5} {pidx:>4} {w[0]:>4} {w[1]:>5} {ul:>10}")
|
||||
|
||||
walls_py = [None if w is None or w[0] is None or w[1] is None else tuple(w)
|
||||
for w in walls]
|
||||
bv = py.estimate_bladder_volume_6ch(walls_py)
|
||||
if bv:
|
||||
print(f"\nPython BV: {bv.volume_ml:.1f} mL lr={bv.lr_ratio:.3f}")
|
||||
else:
|
||||
print("\nPython BV: failed (< 4 center walls or fit failed)")
|
||||
|
||||
|
||||
def main():
|
||||
p = argparse.ArgumentParser(
|
||||
description="VesiScan CSV analyzer",
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog=__doc__)
|
||||
p.add_argument("csv", type=Path, help="App-exported CSV path")
|
||||
p.add_argument("--true", dest="true_vol", type=float, required=True,
|
||||
help="Known true volume (mL)")
|
||||
p.add_argument("--ablation", action="store_true",
|
||||
help="Run 5 algorithm-config comparison")
|
||||
p.add_argument("--plot", type=Path, metavar="DIR",
|
||||
help="Save plots to DIR")
|
||||
p.add_argument("--cycle", type=int, metavar="SID",
|
||||
help="Inspect single scan_id detail")
|
||||
p.add_argument("--appshare", type=str, default=None,
|
||||
help="Path to appshare repo (default: auto-detect)")
|
||||
args = p.parse_args()
|
||||
|
||||
if not args.csv.exists():
|
||||
sys.exit(f"CSV not found: {args.csv}")
|
||||
|
||||
appshare_dir = _find_appshare(args.appshare)
|
||||
sys.path.insert(0, str(appshare_dir))
|
||||
|
||||
# Lazy imports so --help works without numpy etc.
|
||||
import numpy as np # noqa
|
||||
import for_app_share as _fas
|
||||
import vesiscan_test.library.method_d as _md
|
||||
|
||||
class _PyAPI:
|
||||
pass
|
||||
_PyAPI.estimate_bladder_volume_6ch = staticmethod(_fas.estimate_bladder_volume_6ch)
|
||||
_PyAPI.detect_walls_multichannel = staticmethod(_md.detect_walls_multichannel)
|
||||
|
||||
print(f"Loading {args.csv.name}...")
|
||||
scans = load_scans(args.csv)
|
||||
print(f" {len(scans)} scans\n")
|
||||
|
||||
if args.cycle is not None:
|
||||
mode_cycle(scans, args.cycle, _PyAPI)
|
||||
return
|
||||
|
||||
if args.ablation:
|
||||
print("─" * 60); print("ABLATION MODE"); print("─" * 60)
|
||||
mode_ablation(scans, args.true_vol, _PyAPI)
|
||||
print()
|
||||
|
||||
print("─" * 60); print("COMPARE MODE"); print("─" * 60)
|
||||
valid = mode_compare(scans, args.true_vol, _PyAPI)
|
||||
|
||||
if args.plot:
|
||||
print()
|
||||
mode_plot(valid, args.true_vol, args.plot)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user