Skip to content

inputs

topic_segmentation.data.mimic.inputs

Build sentence-level MIMIC inputs for block and subheading segmentation.

python -m topic_segmentation.data.mimic.inputs --segments SEGMENTS.json --out-dir DIR

Write one note per JSONL row to mimic_<table>_<level>.jsonl. Label 1 marks the last sentence of a block, or of a prelude or heading section at subheading level.

add_unit(sentences, labels, unit)

Append non-empty sentences and mark the last as a boundary.

Source code in src/topic_segmentation/data/mimic/inputs.py
15
16
17
18
19
20
def add_unit(sentences, labels, unit):
    """Append non-empty sentences and mark the last as a boundary."""
    unit = [sentence.strip() for sentence in unit if sentence.strip()]
    if unit:
        sentences.extend(unit)
        labels.extend([0] * (len(unit) - 1) + [1])

flatten_note(note, level)

Source code in src/topic_segmentation/data/mimic/inputs.py
23
24
25
26
27
28
29
30
31
32
33
def flatten_note(note, level):
    sentences, labels = [], []
    for block in note["blocks"]:
        prelude = block["prelude"]["sentences"] if block["prelude"] else []
        sections = [section["sentences"] for section in block["sections"]]
        units = [prelude + [sentence for section in sections for sentence in section]] if level == "block" else [prelude, *sections]
        for unit in units:
            add_unit(sentences, labels, unit)
    if labels:
        labels[-1] = 1
    return sentences, labels

rows(notes, level)

Source code in src/topic_segmentation/data/mimic/inputs.py
36
37
38
39
40
41
42
43
44
def rows(notes, level):
    output = []
    for note in notes:
        sentences, labels = flatten_note(note, level)
        if sentences:
            output.append({"articleId": f"{note['note_id']}::{level}", "note_id": note["note_id"], "table": note["table"],
                           "level": level, "sentences": sentences, "labels": labels, "n_sentences": len(sentences),
                           "n_internal_boundaries": max(0, sum(labels) - 1)})
    return output

main()

Source code in src/topic_segmentation/data/mimic/inputs.py
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
def main():
    parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
    parser.add_argument("--segments", type=Path, required=True, help="Segments file of data.mimic.segments")
    parser.add_argument("--out-dir", type=Path, required=True)
    args = parser.parse_args()
    payload = json.loads(args.segments.read_text(encoding="utf-8"))
    args.out_dir.mkdir(parents=True, exist_ok=True)
    for table, note_ids in payload["order"].items():
        notes = [payload["notes"][note_id] for note_id in note_ids]
        for level in LEVELS:
            output = rows(notes, level)
            with (args.out_dir / f"mimic_{table}_{level}.jsonl").open("w", encoding="utf-8") as stream:
                stream.writelines(json.dumps(row, ensure_ascii=False) + "\n" for row in output)
            print(f"{table} {level}: {len(output)} notes, {sum(row['n_sentences'] for row in output)} sentences, "
                  f"{sum(row['n_internal_boundaries'] for row in output)} internal boundaries")