Blocks and heading sections of MIMIC discharge notes, split into sentences.
python -m topic_segmentation.data.mimic.segments --cache NOTES_by_id.json [--cache MORE_by_id.json] --out SEGMENTS.json
Blocks are separated by blank lines; within a block, hash, colon, underlined, and standalone headings open sections
(data.mimic.rules). Lines before the first heading form the block prelude.
start_section(sections: list[dict[str, Any]], current: dict[str, Any] | None, label: str, header_type: str, raw_header: str, block_id: int) -> dict[str, Any]
Source code in src/topic_segmentation/data/mimic/segments.py
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44 | def start_section(
sections: list[dict[str, Any]],
current: dict[str, Any] | None,
label: str,
header_type: str,
raw_header: str,
block_id: int,
) -> dict[str, Any]:
if current is not None:
sections.append(current)
return {
"section_id": f"b{block_id:03d}_s{len(sections):03d}",
"label": label,
"header_type": header_type,
"raw_header": raw_header,
"_lines": [],
}
|
segment_block(block_id: int, block_lines: list[tuple[int, str]]) -> dict[str, Any]
Source code in src/topic_segmentation/data/mimic/segments.py
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105 | def segment_block(block_id: int, block_lines: list[tuple[int, str]]) -> dict[str, Any]:
prelude_lines: list[str] = []
sections: list[dict[str, Any]] = []
current: dict[str, Any] | None = None
i = 0
while i < len(block_lines):
_, line = block_lines[i]
if i + 1 < len(block_lines):
underline_label = underline_heading_label(line, block_lines[i + 1][1])
if underline_label is not None:
current = start_section(
sections,
current,
underline_label,
"underline_heading",
f"{line.strip()}\n{block_lines[i + 1][1].strip()}",
block_id,
)
i += 2
continue
header, header_type = hash_header_parts(line), "hash_heading"
if header is None:
header = colon_header_parts(line)
header_type = "major_inline_heading" if header is not None and header[1] else "colon_heading"
if header is not None:
label, value = header
# A repeated heading right after the same heading continues that section.
repeated = current is not None and label_key(label) == label_key(current["label"]) and not current["_lines"]
if not repeated:
current = start_section(sections, current, label, header_type, line.strip(), block_id)
if value:
current["_lines"].append(value)
underlined = (not repeated and header_type != "hash_heading" and i + 1 < len(block_lines)
and UNDERLINE_RE.match(block_lines[i + 1][1]))
i += 2 if underlined else 1
continue
standalone_header = standalone_header_parts(line)
if standalone_header is not None:
label, _ = standalone_header
current = start_section(sections, current, label, "standalone_heading", line.strip(), block_id)
i += 1
continue
if current is None:
prelude_lines.append(line)
else:
current["_lines"].append(line)
i += 1
if current is not None:
sections.append(current)
for section in sections:
section["sentences"] = split_sentences(section.pop("_lines"))
prelude = split_sentences(prelude_lines)
return {"block_id": block_id, "prelude": {"sentences": prelude} if prelude else None, "sections": sections}
|
build_note_segments(note)
Normalize front matter and split a note into blocks, preludes, and heading sections.
Source code in src/topic_segmentation/data/mimic/segments.py
| def build_note_segments(note):
"""Normalize front matter and split a note into blocks, preludes, and heading sections."""
note = normalize_front_matter_doc(note)
blocks = [segment_block(block_id, lines) for block_id, (_, _, lines) in enumerate(split_blocks(note["text"]))]
return {"note_id": note["note_id"], "table": note["table"], "subject_id": note["subject_id"],
"hadm_id": note["hadm_id"], "blocks": blocks}
|
main()
Source code in src/topic_segmentation/data/mimic/segments.py
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132 | def main():
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--cache", type=Path, action="append", required=True,
help="Note cache (*_by_id.json) of data.mimic.sample; repeat for several")
parser.add_argument("--out", type=Path, required=True)
args = parser.parse_args()
order, notes = {}, {}
for path in args.cache:
cache = json.loads(path.read_text(encoding="utf-8"))
for note_id in cache["order"]:
if note_id in notes:
raise ValueError(f"note {note_id} is in two caches")
notes[note_id] = build_note_segments(cache["notes"][note_id])
order.setdefault(notes[note_id]["table"], []).append(note_id)
args.out.parent.mkdir(parents=True, exist_ok=True)
args.out.write_text(json.dumps({"order": order, "notes": notes}, ensure_ascii=False, indent=2), encoding="utf-8")
print(f"wrote {len(notes)} notes to {args.out}")
|