Build sentence-level MIMIC inputs for block and subheading segmentation.
python -m topic_segmentation.data.mimic.inputs --segments SEGMENTS.json --out-dir DIR
Write one note per JSONL row to mimic_<table>_<level>.jsonl. Label 1 marks the
last sentence of a block, or of a prelude or heading section at subheading level.
Append non-empty sentences and mark the last as a boundary.
Source code in src/topic_segmentation/data/mimic/inputs.py
| def add_unit(sentences, labels, unit):
"""Append non-empty sentences and mark the last as a boundary."""
unit = [sentence.strip() for sentence in unit if sentence.strip()]
if unit:
sentences.extend(unit)
labels.extend([0] * (len(unit) - 1) + [1])
|
Source code in src/topic_segmentation/data/mimic/inputs.py
23
24
25
26
27
28
29
30
31
32
33 | def flatten_note(note, level):
sentences, labels = [], []
for block in note["blocks"]:
prelude = block["prelude"]["sentences"] if block["prelude"] else []
sections = [section["sentences"] for section in block["sections"]]
units = [prelude + [sentence for section in sections for sentence in section]] if level == "block" else [prelude, *sections]
for unit in units:
add_unit(sentences, labels, unit)
if labels:
labels[-1] = 1
return sentences, labels
|
Source code in src/topic_segmentation/data/mimic/inputs.py
36
37
38
39
40
41
42
43
44 | def rows(notes, level):
output = []
for note in notes:
sentences, labels = flatten_note(note, level)
if sentences:
output.append({"articleId": f"{note['note_id']}::{level}", "note_id": note["note_id"], "table": note["table"],
"level": level, "sentences": sentences, "labels": labels, "n_sentences": len(sentences),
"n_internal_boundaries": max(0, sum(labels) - 1)})
return output
|
main()
Source code in src/topic_segmentation/data/mimic/inputs.py
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61 | def main():
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--segments", type=Path, required=True, help="Segments file of data.mimic.segments")
parser.add_argument("--out-dir", type=Path, required=True)
args = parser.parse_args()
payload = json.loads(args.segments.read_text(encoding="utf-8"))
args.out_dir.mkdir(parents=True, exist_ok=True)
for table, note_ids in payload["order"].items():
notes = [payload["notes"][note_id] for note_id in note_ids]
for level in LEVELS:
output = rows(notes, level)
with (args.out_dir / f"mimic_{table}_{level}.jsonl").open("w", encoding="utf-8") as stream:
stream.writelines(json.dumps(row, ensure_ascii=False) + "\n" for row in output)
print(f"{table} {level}: {len(output)} notes, {sum(row['n_sentences'] for row in output)} sentences, "
f"{sum(row['n_internal_boundaries'] for row in output)} internal boundaries")
|