Collect MIMIC scaling scores and paired significance tests.
python -m topic_segmentation.clinical_statistics
Read fine-tuning runs from --scaling and zero-shot predictions from --zero-shot.
Defaults come from configs/mimic_scaling.json and configs/mimic_inference.json.
Write scaling.csv and t_tests.json to <scaling>/statistics/. Scores are percentages;
training_notes=0 denotes zero-shot results. Use one-sided paired t-tests of per-note
boundary F1, and rescore zero-shot predictions with the shared metrics.
run_dir(scaling, model, level, size)
Return the run directory for one initialization, level, and training size.
Source code in src/topic_segmentation/clinical_statistics.py
| def run_dir(scaling, model, level, size):
"""Return the run directory for one initialization, level, and training size."""
return scaling / INITIALIZATIONS[model] / level / f"n{size:04d}" / "ts_noda" / f"seed{SCALING['seed']}"
|
note_scores(scaling, model, level, size)
Return note keys, gold labels, and per-note boundary F1 in percent.
Source code in src/topic_segmentation/clinical_statistics.py
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58 | def note_scores(scaling, model, level, size):
"""Return note keys, gold labels, and per-note boundary F1 in percent."""
[path] = run_dir(scaling, model, level, size).glob("predict_*_ts_score_lt.txt")
identities, labels, values = [], [], []
for row in map(json.loads, path.open()):
gold, predicted = row["labels"], row["predictions"]
assert len(gold) == len(predicted)
# Identify notes by their opening units, before overlapping windows repeat content.
identities.append(json.dumps(row["sentences"][0][:20], ensure_ascii=False))
labels.append(gold)
tp = sum(g == p == "B-EOP" for g, p in zip(gold, predicted))
errors = sum((g == "B-EOP") != (p == "B-EOP") for g, p in zip(gold, predicted))
values.append(200 * tp / (2 * tp + errors) if 2 * tp + errors else 0.0)
assert len(identities) == len(set(identities)) == 1000
return identities, labels, np.array(values)
|
t_tests(scaling, output)
Source code in src/topic_segmentation/clinical_statistics.py
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75 | def t_tests(scaling, output):
results = []
for level in LEVELS:
for size in SIZES:
scores = {model: note_scores(scaling, model, level, size) for model in INITIALIZATIONS}
assert all(identities == scores["Base"][0] and labels == scores["Base"][1]
for identities, labels, _ in scores.values())
for left, right in COMPARISONS:
x, y = scores[left][2], scores[right][2]
statistic, p = (0.0, 1.0) if np.all(x == y) else ttest_rel(x, y, alternative="greater")
results.append(dict(task=level, training_notes=size, n=len(x), left=left, right=right,
mean_difference=float((x - y).mean()), t=float(statistic), p=float(p)))
output = output / "t_tests.json"
output.write_text(json.dumps(results, indent=2, allow_nan=False) + "\n", encoding="utf-8")
print(f"wrote {output}")
|
zero_shot_scores(zero_shot, level, prefix)
Score zero-shot predictions against the gold notes and return their path and scores.
Source code in src/topic_segmentation/clinical_statistics.py
78
79
80
81
82
83
84
85
86 | def zero_shot_scores(zero_shot, level, prefix):
"""Score zero-shot predictions against the gold notes and return their path and scores."""
with (TEST_INPUTS / f"mimic_discharge_{level}.jsonl").open(encoding="utf-8") as stream:
gold = {row["articleId"]: row for row in map(json.loads, stream)}
path = zero_shot / f"{prefix}_discharge_{level}.jsonl"
with path.open(encoding="utf-8") as stream:
rows = [json.loads(line) for line in stream]
assert len(rows) == len(gold)
return path, score_articles([gold[row["articleId"]] for row in rows], [row["predictions"] for row in rows])
|
collect(scaling, zero_shot, output)
Source code in src/topic_segmentation/clinical_statistics.py
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110 | def collect(scaling, zero_shot, output):
table = []
def add(level, model, size, path, values):
scores = {name: 100.0 * float(values[name]) for name in METRICS}
table.append({"level": level, "model": model, "training_notes": size,
"source": os.path.relpath(path, REPOSITORY), **scores})
for level in LEVELS:
for size in SIZES:
for model in INITIALIZATIONS:
path = run_dir(scaling, model, level, size) / "all_results.json"
result = json.loads(path.read_text(encoding="utf-8"))
add(level, model, size, path, {name: result[f"threshold_0.5_example_level_{name}"] for name in METRICS})
for model, prefix in ZERO_SHOT.items():
add(level, model, 0, *zero_shot_scores(zero_shot, level, prefix))
output = output / "scaling.csv"
with output.open("w", encoding="utf-8", newline="") as stream:
writer = csv.DictWriter(stream, fieldnames=list(table[0]))
writer.writeheader()
writer.writerows(table)
print(f"wrote {len(table)} rows to {output}")
|
main()
Source code in src/topic_segmentation/clinical_statistics.py
113
114
115
116
117
118
119
120
121
122 | def main():
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--scaling", type=Path, default=REPOSITORY / SCALING["output"], help="Fine-tuning runs")
parser.add_argument("--zero-shot", type=Path, default=REPOSITORY / INFERENCE["output"],
help="Zero-shot predictions")
args = parser.parse_args()
output = args.scaling / "statistics"
output.mkdir(parents=True, exist_ok=True)
collect(args.scaling.resolve(), args.zero_shot.resolve(), output)
t_tests(args.scaling.resolve(), output)
|