Skip to content

clinical_statistics

topic_segmentation.clinical_statistics

Collect MIMIC scaling scores and paired significance tests.

python -m topic_segmentation.clinical_statistics

Read fine-tuning runs from --scaling and zero-shot predictions from --zero-shot. Defaults come from configs/mimic_scaling.json and configs/mimic_inference.json. Write scaling.csv and t_tests.json to <scaling>/statistics/. Scores are percentages; training_notes=0 denotes zero-shot results. Use one-sided paired t-tests of per-note boundary F1, and rescore zero-shot predictions with the shared metrics.

run_dir(scaling, model, level, size)

Return the run directory for one initialization, level, and training size.

Source code in src/topic_segmentation/clinical_statistics.py
39
40
41
def run_dir(scaling, model, level, size):
    """Return the run directory for one initialization, level, and training size."""
    return scaling / INITIALIZATIONS[model] / level / f"n{size:04d}" / "ts_noda" / f"seed{SCALING['seed']}"

note_scores(scaling, model, level, size)

Return note keys, gold labels, and per-note boundary F1 in percent.

Source code in src/topic_segmentation/clinical_statistics.py
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
def note_scores(scaling, model, level, size):
    """Return note keys, gold labels, and per-note boundary F1 in percent."""
    [path] = run_dir(scaling, model, level, size).glob("predict_*_ts_score_lt.txt")
    identities, labels, values = [], [], []
    for row in map(json.loads, path.open()):
        gold, predicted = row["labels"], row["predictions"]
        assert len(gold) == len(predicted)
        # Identify notes by their opening units, before overlapping windows repeat content.
        identities.append(json.dumps(row["sentences"][0][:20], ensure_ascii=False))
        labels.append(gold)
        tp = sum(g == p == "B-EOP" for g, p in zip(gold, predicted))
        errors = sum((g == "B-EOP") != (p == "B-EOP") for g, p in zip(gold, predicted))
        values.append(200 * tp / (2 * tp + errors) if 2 * tp + errors else 0.0)
    assert len(identities) == len(set(identities)) == 1000
    return identities, labels, np.array(values)

t_tests(scaling, output)

Source code in src/topic_segmentation/clinical_statistics.py
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
def t_tests(scaling, output):
    results = []
    for level in LEVELS:
        for size in SIZES:
            scores = {model: note_scores(scaling, model, level, size) for model in INITIALIZATIONS}
            assert all(identities == scores["Base"][0] and labels == scores["Base"][1]
                       for identities, labels, _ in scores.values())
            for left, right in COMPARISONS:
                x, y = scores[left][2], scores[right][2]
                statistic, p = (0.0, 1.0) if np.all(x == y) else ttest_rel(x, y, alternative="greater")
                results.append(dict(task=level, training_notes=size, n=len(x), left=left, right=right,
                                    mean_difference=float((x - y).mean()), t=float(statistic), p=float(p)))
    output = output / "t_tests.json"
    output.write_text(json.dumps(results, indent=2, allow_nan=False) + "\n", encoding="utf-8")
    print(f"wrote {output}")

zero_shot_scores(zero_shot, level, prefix)

Score zero-shot predictions against the gold notes and return their path and scores.

Source code in src/topic_segmentation/clinical_statistics.py
78
79
80
81
82
83
84
85
86
def zero_shot_scores(zero_shot, level, prefix):
    """Score zero-shot predictions against the gold notes and return their path and scores."""
    with (TEST_INPUTS / f"mimic_discharge_{level}.jsonl").open(encoding="utf-8") as stream:
        gold = {row["articleId"]: row for row in map(json.loads, stream)}
    path = zero_shot / f"{prefix}_discharge_{level}.jsonl"
    with path.open(encoding="utf-8") as stream:
        rows = [json.loads(line) for line in stream]
    assert len(rows) == len(gold)
    return path, score_articles([gold[row["articleId"]] for row in rows], [row["predictions"] for row in rows])

collect(scaling, zero_shot, output)

Source code in src/topic_segmentation/clinical_statistics.py
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
def collect(scaling, zero_shot, output):
    table = []

    def add(level, model, size, path, values):
        scores = {name: 100.0 * float(values[name]) for name in METRICS}
        table.append({"level": level, "model": model, "training_notes": size,
                      "source": os.path.relpath(path, REPOSITORY), **scores})

    for level in LEVELS:
        for size in SIZES:
            for model in INITIALIZATIONS:
                path = run_dir(scaling, model, level, size) / "all_results.json"
                result = json.loads(path.read_text(encoding="utf-8"))
                add(level, model, size, path, {name: result[f"threshold_0.5_example_level_{name}"] for name in METRICS})
        for model, prefix in ZERO_SHOT.items():
            add(level, model, 0, *zero_shot_scores(zero_shot, level, prefix))
    output = output / "scaling.csv"
    with output.open("w", encoding="utf-8", newline="") as stream:
        writer = csv.DictWriter(stream, fieldnames=list(table[0]))
        writer.writeheader()
        writer.writerows(table)
    print(f"wrote {len(table)} rows to {output}")

main()

Source code in src/topic_segmentation/clinical_statistics.py
113
114
115
116
117
118
119
120
121
122
def main():
    parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
    parser.add_argument("--scaling", type=Path, default=REPOSITORY / SCALING["output"], help="Fine-tuning runs")
    parser.add_argument("--zero-shot", type=Path, default=REPOSITORY / INFERENCE["output"],
                        help="Zero-shot predictions")
    args = parser.parse_args()
    output = args.scaling / "statistics"
    output.mkdir(parents=True, exist_ok=True)
    collect(args.scaling.resolve(), args.zero_shot.resolve(), output)
    t_tests(args.scaling.resolve(), output)