Skip to content

segments

topic_segmentation.data.mimic.segments

Blocks and heading sections of MIMIC discharge notes, split into sentences.

python -m topic_segmentation.data.mimic.segments --cache NOTES_by_id.json [--cache MORE_by_id.json] --out SEGMENTS.json

Blocks are separated by blank lines; within a block, hash, colon, underlined, and standalone headings open sections (data.mimic.rules). Lines before the first heading form the block prelude.

start_section(sections: list[dict[str, Any]], current: dict[str, Any] | None, label: str, header_type: str, raw_header: str, block_id: int) -> dict[str, Any]

Source code in src/topic_segmentation/data/mimic/segments.py
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
def start_section(
    sections: list[dict[str, Any]],
    current: dict[str, Any] | None,
    label: str,
    header_type: str,
    raw_header: str,
    block_id: int,
) -> dict[str, Any]:
    if current is not None:
        sections.append(current)
    return {
        "section_id": f"b{block_id:03d}_s{len(sections):03d}",
        "label": label,
        "header_type": header_type,
        "raw_header": raw_header,
        "_lines": [],
    }

segment_block(block_id: int, block_lines: list[tuple[int, str]]) -> dict[str, Any]

Source code in src/topic_segmentation/data/mimic/segments.py
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
def segment_block(block_id: int, block_lines: list[tuple[int, str]]) -> dict[str, Any]:
    prelude_lines: list[str] = []
    sections: list[dict[str, Any]] = []
    current: dict[str, Any] | None = None

    i = 0
    while i < len(block_lines):
        _, line = block_lines[i]
        if i + 1 < len(block_lines):
            underline_label = underline_heading_label(line, block_lines[i + 1][1])
            if underline_label is not None:
                current = start_section(
                    sections,
                    current,
                    underline_label,
                    "underline_heading",
                    f"{line.strip()}\n{block_lines[i + 1][1].strip()}",
                    block_id,
                )
                i += 2
                continue

        header, header_type = hash_header_parts(line), "hash_heading"
        if header is None:
            header = colon_header_parts(line)
            header_type = "major_inline_heading" if header is not None and header[1] else "colon_heading"
        if header is not None:
            label, value = header
            # A repeated heading right after the same heading continues that section.
            repeated = current is not None and label_key(label) == label_key(current["label"]) and not current["_lines"]
            if not repeated:
                current = start_section(sections, current, label, header_type, line.strip(), block_id)
            if value:
                current["_lines"].append(value)
            underlined = (not repeated and header_type != "hash_heading" and i + 1 < len(block_lines)
                          and UNDERLINE_RE.match(block_lines[i + 1][1]))
            i += 2 if underlined else 1
            continue

        standalone_header = standalone_header_parts(line)
        if standalone_header is not None:
            label, _ = standalone_header
            current = start_section(sections, current, label, "standalone_heading", line.strip(), block_id)
            i += 1
            continue

        if current is None:
            prelude_lines.append(line)
        else:
            current["_lines"].append(line)
        i += 1

    if current is not None:
        sections.append(current)

    for section in sections:
        section["sentences"] = split_sentences(section.pop("_lines"))
    prelude = split_sentences(prelude_lines)
    return {"block_id": block_id, "prelude": {"sentences": prelude} if prelude else None, "sections": sections}

build_note_segments(note)

Normalize front matter and split a note into blocks, preludes, and heading sections.

Source code in src/topic_segmentation/data/mimic/segments.py
108
109
110
111
112
113
def build_note_segments(note):
    """Normalize front matter and split a note into blocks, preludes, and heading sections."""
    note = normalize_front_matter_doc(note)
    blocks = [segment_block(block_id, lines) for block_id, (_, _, lines) in enumerate(split_blocks(note["text"]))]
    return {"note_id": note["note_id"], "table": note["table"], "subject_id": note["subject_id"],
            "hadm_id": note["hadm_id"], "blocks": blocks}

main()

Source code in src/topic_segmentation/data/mimic/segments.py
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
def main():
    parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
    parser.add_argument("--cache", type=Path, action="append", required=True,
                        help="Note cache (*_by_id.json) of data.mimic.sample; repeat for several")
    parser.add_argument("--out", type=Path, required=True)
    args = parser.parse_args()
    order, notes = {}, {}
    for path in args.cache:
        cache = json.loads(path.read_text(encoding="utf-8"))
        for note_id in cache["order"]:
            if note_id in notes:
                raise ValueError(f"note {note_id} is in two caches")
            notes[note_id] = build_note_segments(cache["notes"][note_id])
            order.setdefault(notes[note_id]["table"], []).append(note_id)
    args.out.parent.mkdir(parents=True, exist_ok=True)
    args.out.write_text(json.dumps({"order": order, "notes": notes}, ensure_ascii=False, indent=2), encoding="utf-8")
    print(f"wrote {len(notes)} notes to {args.out}")