Rules for splitting MIMIC notes into front matter, blocks, sections, and sentences.
split_front_matter_line(line: str) -> list[str]
Split a two-column front-matter line into fields; keep other lines as a single field.
Source code in src/topic_segmentation/data/mimic/rules.py
110
111
112
113
114
115
116 | def split_front_matter_line(line: str) -> list[str]:
"""Split a two-column front-matter line into fields; keep other lines as a single field."""
match = RIGHT_FIELD_RE.match(line)
left = match.group("left").rstrip() if match else ""
if not match or not any(left.lstrip().startswith(f"{label}:") for label in FIELD_SPLITS[match.group("label")]):
return [line]
return [left, match.group("right").lstrip()]
|
normalize_front_matter_doc(doc: dict, max_front_matter_lines: int = 30) -> dict
Split two-column front matter into separate lines before the first body heading.
Source code in src/topic_segmentation/data/mimic/rules.py
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134 | def normalize_front_matter_doc(doc: dict, max_front_matter_lines: int = 30) -> dict:
"""Split two-column front matter into separate lines before the first body heading."""
out, in_front_matter = [], True
for line_index, raw_line in enumerate(doc["text"].splitlines(keepends=True)):
line = raw_line.rstrip("\r\n")
linebreak = raw_line[len(line):]
if in_front_matter and BODY_START_RE.match(line):
in_front_matter = False
fields = split_front_matter_line(line) if in_front_matter and line_index < max_front_matter_lines else [line]
if len(fields) == 1:
out.append(raw_line)
elif linebreak:
out.extend(f"{field}{linebreak}" for field in fields)
else:
out.append("\n".join(fields))
return dict(doc, text="".join(out))
|
label_key(label: str) -> str
Source code in src/topic_segmentation/data/mimic/rules.py
| def label_key(label: str) -> str:
return re.sub(r"[^A-Z0-9]+", " ", label.upper()).strip()
|
clean_label(label: str) -> str
Source code in src/topic_segmentation/data/mimic/rules.py
141
142
143
144
145
146
147
148
149 | def clean_label(label: str) -> str:
label = " ".join(label.split())
label = label.strip("* ")
label = re.sub(r"^\s*[-*•]\s*", "", label).strip()
raw_key = label.upper()
if raw_key in SPECIAL_LABEL_ALIASES:
return SPECIAL_LABEL_ALIASES[raw_key]
key = label_key(label)
return SPECIAL_LABEL_ALIASES.get(key, label)
|
is_major_label(label: str) -> bool
Source code in src/topic_segmentation/data/mimic/rules.py
| def is_major_label(label: str) -> bool:
return label_key(clean_label(label)) in MAJOR_SECTION_LABELS
|
looks_like_sentence_label(label: str) -> bool
Source code in src/topic_segmentation/data/mimic/rules.py
156
157
158
159
160
161
162
163
164
165 | def looks_like_sentence_label(label: str) -> bool:
key = f" {label_key(label)} "
words = key.strip().split()
if "." in label:
return True
if len(words) >= 2 and words[0] in {"HE", "SHE", "PATIENT", "PATIENTS", "THEY", "IT", "WE"}:
return True
if len(words) >= 4 and words[0] in SENTENCE_LABEL_STARTS:
return True
return any(phrase in key for phrase in SENTENCE_LABEL_PHRASES)
|
looks_like_medication_label(label: str) -> bool
Source code in src/topic_segmentation/data/mimic/rules.py
168
169
170
171
172
173
174
175
176
177
178 | def looks_like_medication_label(label: str) -> bool:
key = label_key(label)
if key in {"SIG", "START", "DISP", "DURATION", "REASON FOR PRN DUPLICATE OVERRIDE"}:
return True
if "REASON FOR PRN DUPLICATE OVERRIDE" in key or "REASON FOR ORDERING" in key:
return True
if key.startswith(("SUBCUTANEOUS", "INTRAVENOUS", "INTRAMUSCULAR", "NEBULIZATION", "TABLET", "CAPSULE")):
return True
if key.startswith("PICC ") or " HEPARIN DEPENDENT" in key:
return True
return bool(MEDICATION_LABEL_RE.search(label))
|
label_ok(label: str, *, allow_major: bool = False, allow_problem: bool = False) -> bool
Source code in src/topic_segmentation/data/mimic/rules.py
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203 | def label_ok(label: str, *, allow_major: bool = False, allow_problem: bool = False) -> bool:
label = clean_label(label)
key = label_key(label)
letters = sum(ch.isalpha() for ch in label)
digits = sum(ch.isdigit() for ch in label)
if letters < 2 or digits > max(4, letters * 2):
return False
if any(ch in label for ch in "()[]{}"):
return False
if "." in label:
return False
if "," in label and len(label) > 24 and label_key(label) not in MAJOR_SECTION_LABELS:
return False
first_alpha = next((ch for ch in label if ch.isalpha()), "")
if first_alpha and first_alpha.islower():
return False
if key in REJECT_ONE_WORD_LABELS:
return False
if not allow_problem and looks_like_sentence_label(label):
return False
if not allow_major and looks_like_medication_label(label):
return False
return True
|
normalized_heading_line(line: str) -> str
Source code in src/topic_segmentation/data/mimic/rules.py
| def normalized_heading_line(line: str) -> str:
stripped = line.strip()
if stripped.startswith("=") and stripped.endswith("="):
stripped = stripped.strip("= ").strip()
return stripped
|
split_hash_label_value(text: str) -> tuple[str, str | None]
Source code in src/topic_segmentation/data/mimic/rules.py
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230 | def split_hash_label_value(text: str) -> tuple[str, str | None]:
for index, char in enumerate(text):
if char != ":":
continue
prev_char = text[index - 1] if index else ""
next_char = text[index + 1] if index + 1 < len(text) else ""
if prev_char.isdigit() and next_char.isdigit():
continue
if next_char and not next_char.isspace():
continue
return text[:index].strip(), text[index + 1 :].strip() or None
dash_match = re.search(r"\s*-\s+", text)
if dash_match:
label = text[: dash_match.start()].strip()
value = text[dash_match.end() :].strip()
if 1 <= len(label.split()) <= 8 and value:
return label.rstrip(".").strip(), value
return text.rstrip(".").strip(), None
|
Source code in src/topic_segmentation/data/mimic/rules.py
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249 | def hash_header_parts(line: str) -> tuple[str, str | None] | None:
stripped = line.strip()
if not stripped.startswith("#"):
return None
after_hash = stripped.lstrip("#").lstrip()
if after_hash.startswith("*"):
return None
after_hash = re.sub(r"^[).]\s*", "", after_hash).strip()
label_text, value = split_hash_label_value(after_hash)
label = clean_label(label_text)
if "REFILL" in label_key(label):
return None
if value is None and len(label.split()) > 8:
return None
if not label_ok(label, allow_problem=True):
return None
return label, value
|
Source code in src/topic_segmentation/data/mimic/rules.py
252
253
254
255
256
257
258
259
260
261
262
263
264 | def colon_header_parts(line: str) -> tuple[str, str | None] | None:
match = COLON_HEADER_RE.match(normalized_heading_line(line))
if not match:
return None
label = clean_label(match.group("label"))
value = match.group("value").strip()
if label_key(label) in FRONT_MATTER_LABELS and not value:
return None
if value:
if is_major_label(label) and label_ok(label, allow_major=True):
return label, value
return None
return (label, None) if label_ok(label) else None
|
subsection_parts(line: str) -> tuple[str, str | None] | None
Source code in src/topic_segmentation/data/mimic/rules.py
| def subsection_parts(line: str) -> tuple[str, str | None] | None:
return hash_header_parts(line) or colon_header_parts(line) or standalone_header_parts(line)
|
standalone_header_parts(line: str) -> tuple[str, str | None] | None
Source code in src/topic_segmentation/data/mimic/rules.py
271
272
273
274
275
276
277
278
279
280 | def standalone_header_parts(line: str) -> tuple[str, str | None] | None:
label = clean_label(normalized_heading_line(line))
if not label or ":" in label:
return None
key = label_key(label)
if key not in MAJOR_SECTION_LABELS:
return None
if not label_ok(label, allow_major=True):
return None
return label, None
|
underline_heading_label(line: str, next_line: str) -> str | None
Source code in src/topic_segmentation/data/mimic/rules.py
283
284
285
286
287
288
289
290
291
292
293
294
295 | def underline_heading_label(line: str, next_line: str) -> str | None:
if not UNDERLINE_RE.match(next_line):
return None
label = clean_label(line.strip())
if not label or ":" in label:
return None
key = label_key(label)
is_upper_heading = any(ch.isalpha() for ch in label) and label.upper() == label
if not is_upper_heading and key not in MAJOR_SECTION_LABELS:
return None
if key in FRONT_MATTER_LABELS:
return None
return label if label_ok(label, allow_major=True) else None
|
split_blocks(text: str) -> list[tuple[int, int, list[tuple[int, str]]]]
Source code in src/topic_segmentation/data/mimic/rules.py
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324 | def split_blocks(text: str) -> list[tuple[int, int, list[tuple[int, str]]]]:
blocks: list[tuple[int, int, list[tuple[int, str]]]] = []
current: list[tuple[int, str]] = []
pending_blanks: list[tuple[int, str]] = []
start_line: int | None = None
for line_no, line in enumerate(text.splitlines(), start=1):
if not line.strip():
if start_line is not None:
pending_blanks.append((line_no, line))
continue
if len(pending_blanks) >= 2:
if current:
blocks.append((start_line or current[0][0], current[-1][0], current))
current = [(line_no, line)]
start_line = line_no
else:
if start_line is None:
start_line = line_no
current.extend(pending_blanks)
current.append((line_no, line))
pending_blanks = []
if current:
blocks.append((start_line or current[0][0], current[-1][0], current))
return blocks
|
split_subsections(block_lines: list[tuple[int, str]]) -> list[tuple[str | None, int | None, list[tuple[int, str]]]]
Source code in src/topic_segmentation/data/mimic/rules.py
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369 | def split_subsections(block_lines: list[tuple[int, str]]) -> list[tuple[str | None, int | None, list[tuple[int, str]]]]:
sections: list[tuple[str | None, int | None, list[tuple[int, str]]]] = []
current_header: str | None = None
current_header_line: int | None = None
current_lines: list[tuple[int, str]] = []
i = 0
while i < len(block_lines):
line_no, line = block_lines[i]
if i + 1 < len(block_lines):
underline_label = underline_heading_label(line, block_lines[i + 1][1])
if underline_label is not None:
if current_header is not None or current_lines:
sections.append((current_header, current_header_line, current_lines))
current_header = underline_label
current_header_line = line_no
current_lines = []
i += 2
continue
subsection = subsection_parts(line)
if subsection is not None:
label, value = subsection
if current_header is not None and label_key(label) == label_key(current_header) and not current_lines:
if value:
current_lines.append((line_no, value))
i += 1
continue
if current_header is not None or current_lines:
sections.append((current_header, current_header_line, current_lines))
current_header = label
current_header_line = line_no
current_lines = []
if value:
current_lines.append((line_no, value))
i += 2 if i + 1 < len(block_lines) and UNDERLINE_RE.match(block_lines[i + 1][1]) else 1
continue
current_lines.append((line_no, line))
i += 1
if current_header is not None or current_lines:
sections.append((current_header, current_header_line, current_lines))
return sections
|
protect_sentence_text(text: str) -> str
Source code in src/topic_segmentation/data/mimic/rules.py
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390 | def protect_sentence_text(text: str) -> str:
protected = text
for index, abbreviation in enumerate(ABBREVIATIONS):
protected = protected.replace(abbreviation, abbreviation.replace(".", f"<ABBR{index}>"))
protected = protected.replace("y.o", "y<DOT>o")
protected = re.sub(r"\by\.\s+o\b", "y<DOT>o", protected)
protected = MEDICAL_DOTTED_TERM_RE.sub(r"\1<DOT> ", protected)
protected = UPPER_DOTTED_ABBREVIATION_RE.sub(
lambda match: match.group(0).replace(".", "<DOT>"), protected
)
protected = LOWER_DOTTED_ABBREVIATION_RE.sub(
lambda match: match.group(0).replace(".", "<DOT>"), protected
)
protected = re.sub(r"([a-z]{4,}\.)\s+(?=[a-z]{3,})", r"\1<SPLIT>", protected)
protected = INLINE_NUMBER_MARKER_RE.sub(r"\1\2<NUMDOT> ", protected)
protected = re.sub(r"^(\s*\d+)\.(?=[A-Za-z])", r"\1<NUMDOT> ", protected)
protected = LEADING_NUMBER_MARKER_RE.sub(r"\1<NUMDOT> ", protected)
protected = DECIMAL_POINT_RE.sub("<DECDOT>", protected)
return protected
|
restore_sentence_text(text: str) -> str
Source code in src/topic_segmentation/data/mimic/rules.py
| def restore_sentence_text(text: str) -> str:
restored = text.replace("<NUMDOT>", ".").replace("<DECDOT>", ".").replace("<DOT>", ".")
for index, abbreviation in enumerate(ABBREVIATIONS):
restored = restored.replace(abbreviation.replace(".", f"<ABBR{index}>"), abbreviation)
return restored
|
split_paragraph(paragraph: str) -> list[str]
Source code in src/topic_segmentation/data/mimic/rules.py
400
401
402
403
404
405
406
407
408 | def split_paragraph(paragraph: str) -> list[str]:
protected = protect_sentence_text(paragraph)
parts = []
for part in SENTENCE_SPLIT_RE.split(protected):
restored = restore_sentence_text(part).strip()
restored = re.sub(r"(?:\s*\.){2,}$", ".", restored)
restored = re.sub(r"\s+\.$", ".", restored)
parts.append(restored)
return [part for part in parts if part and part not in {".", "·"} and not SEPARATOR_RE.match(part)]
|
join_wrapped_lines(lines: list[str]) -> str
Source code in src/topic_segmentation/data/mimic/rules.py
411
412
413
414
415
416
417
418
419
420
421 | def join_wrapped_lines(lines: list[str]) -> str:
joined = ""
for line in lines:
stripped = line.strip()
if not stripped:
continue
if not joined:
joined = stripped
continue
joined += stripped if should_join_hyphenated_wrap(joined, stripped) else f" {stripped}"
return joined
|
should_join_hyphenated_wrap(previous: str, current: str) -> bool
Source code in src/topic_segmentation/data/mimic/rules.py
424
425
426
427
428
429
430
431
432
433
434
435 | def should_join_hyphenated_wrap(previous: str, current: str) -> bool:
if not re.search(r"[A-Za-z]-\s*$", previous):
return False
if re.match(r"(?i)(?:and|or|I:|II:|III:|IV:|V:|VI:|Alert\b)", current):
return False
previous_word = re.search(r"([A-Za-z]+)-\s*$", previous)
current_word = re.match(r"([A-Za-z]+)", current)
if not previous_word or not current_word:
return False
if current_word.group(1)[0].islower():
return previous_word.group(1)[-1].islower() or previous_word.group(1).isupper()
return False
|
preserve_as_own_sentence(line: str) -> bool
Source code in src/topic_segmentation/data/mimic/rules.py
| def preserve_as_own_sentence(line: str) -> bool:
if ":" not in line:
return False
label, value = line.split(":", 1)
return label_key(label) in PRESERVE_FIELD_LABELS and bool(value.strip())
|
number_continues_previous_line(paragraph: list[str], line: str) -> bool
Source code in src/topic_segmentation/data/mimic/rules.py
| def number_continues_previous_line(paragraph: list[str], line: str) -> bool:
return bool(paragraph and re.match(r"^\s*\d+\.\s+", line) and NUMBER_CONTINUATION_PREVIOUS_RE.search(paragraph[-1].strip()))
|
split_sentences(lines: list[str]) -> list[str]
Source code in src/topic_segmentation/data/mimic/rules.py
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490 | def split_sentences(lines: list[str]) -> list[str]:
sentences: list[str] = []
paragraph: list[str] = []
def add_parts(parts: list[str]) -> None:
for part in parts:
if sentences and sentences[-1] in BARE_ABBREVIATIONS:
sentences[-1] = f"{sentences[-1]} {part}"
else:
sentences.append(part)
def flush() -> None:
if paragraph:
add_parts(split_paragraph(join_wrapped_lines(paragraph)))
paragraph.clear()
for line in lines:
stripped = line.strip()
if not stripped:
flush()
continue
if stripped in {".", "·"} or SEPARATOR_RE.match(stripped):
if stripped == "___" and paragraph and paragraph[-1].rstrip().endswith("Dr."):
paragraph.append(stripped)
continue
flush()
continue
if preserve_as_own_sentence(stripped):
flush()
add_parts([stripped])
continue
if paragraph and re.search(r"\by\.$", paragraph[-1].strip()) and re.match(r"(?i)^o\b", stripped):
paragraph.append(stripped)
continue
if LIST_ITEM_RE.match(stripped) and not number_continues_previous_line(paragraph, stripped):
flush()
paragraph.append(stripped)
continue
paragraph.append(stripped)
flush()
return merge_sentence_fragments(sentences)
|
merge_sentence_fragments(sentences: list[str]) -> list[str]
Source code in src/topic_segmentation/data/mimic/rules.py
493
494
495
496
497
498
499
500
501
502
503 | def merge_sentence_fragments(sentences: list[str]) -> list[str]:
merged: list[str] = []
i = 0
while i < len(sentences):
current = sentences[i]
i += 1
while i < len(sentences) and should_merge_with_next(current, sentences[i]):
current = f"{current} {sentences[i]}"
i += 1
merged.append(current)
return merged
|
should_merge_with_next(current: str, next_sentence: str) -> bool
Source code in src/topic_segmentation/data/mimic/rules.py
506
507
508
509
510
511
512
513
514
515
516
517
518
519 | def should_merge_with_next(current: str, next_sentence: str) -> bool:
if re.fullmatch(r"\d+[.)]", current):
return True
if re.search(r"\bSig:\s+\S+\s+\(\d+\)$", current):
return True
if re.search(r"\b(?:Sust|Ophth)\.$|\(E\.C\.\)$", current):
return True
if re.search(r"\b(?:every|times a|once a|by mouth every)$", current, re.IGNORECASE):
return True
if re.search(r"\(\d+$", current) and re.match(r"(?i)^(?:times?|day|hours?|tablet|capsule|ml|neb)\b", next_sentence):
return True
if current.count("(") > current.count(")") and re.fullmatch(r"[\d,;: ]+\)\.?", next_sentence):
return True
return False
|