Prepare MIND sentences, silver labels, and training splits.
python -m topic_segmentation.data.build_mind_silver_labels sentences
python -m topic_segmentation.data.build_mind_silver_labels subsets
python -m topic_segmentation.data.build_mind_silver_labels training-data
scripts/prepare_data.py mind runs the news teacher between sentences and subsets.
Settings come from configs/mind_adaptation.json. Subsets are prefixes of one shuffle;
the test set is the IDEA-seg news test set.
clean_body(text)
Source code in src/topic_segmentation/data/build_mind_silver_labels.py
41
42
43
44
45
46
47
48
49 | def clean_body(text):
text = text.strip()
for pattern in FOOTER_PATTERNS:
match = pattern.search(text)
if match and match.start() > len(text) * 0.5:
text = text[:match.start()]
text = re.sub(r' {2,}', ' ', text)
text = re.sub(r' ([,\.!?;:\)])', r'\1', text)
return text.strip()
|
write_sentences(input_path, output_path)
Clean bodies, split sentences, and keep articles within the sentence limits.
Source code in src/topic_segmentation/data/build_mind_silver_labels.py
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66 | def write_sentences(input_path, output_path):
"""Clean bodies, split sentences, and keep articles within the sentence limits."""
with open(input_path, encoding="utf-8", newline="") as f:
rows = list(csv.DictReader(f))
kept = 0
with open(output_path, "w", encoding="utf-8") as out:
for row in rows:
body = row.get("body", "")
if not body:
continue
sentences = [s.strip() for s in sent_tokenize(clean_body(body)) if len(s.strip()) >= SELECTION["minimum_sentence_characters"]]
if SELECTION["minimum_sentences"] <= len(sentences) <= SELECTION["maximum_sentences"]:
out.write(json.dumps({"articleId": row["news_id"], "sentences": sentences}, ensure_ascii=False) + "\n")
kept += 1
print(f"kept {kept:,} of {len(rows):,} articles -> {output_path}")
|
write_subsets(predictions_path, filtered_path, subsets_dir)
Keep articles within the segment limits, shuffle once, and write nested index files.
Source code in src/topic_segmentation/data/build_mind_silver_labels.py
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84 | def write_subsets(predictions_path, filtered_path, subsets_dir):
"""Keep articles within the segment limits, shuffle once, and write nested index files."""
records = []
with open(predictions_path) as fin, open(filtered_path, "w") as fout:
for line in fin:
d = json.loads(line)
if (SELECTION["minimum_segments"] <= sum(d["predictions"]) <= SELECTION["maximum_segments"]
and d["articleId"] not in EXCLUDED_ARTICLES):
fout.write(line)
records.append({"idx": len(records), "articleId": d["articleId"]})
random.seed(SELECTION["seed"])
random.shuffle(records)
for name, size in SELECTION["subset_sizes"].items():
with open(subsets_dir / f"subset_{name}.json", "w") as f:
json.dump(records[:size], f)
print(f"kept {len(records):,} articles -> {filtered_path}")
|
write_training_data(filtered_path, subsets_dir, news_data, out_root)
Split each subset into train/val; the test split is the news test set.
Source code in src/topic_segmentation/data/build_mind_silver_labels.py
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108 | def write_training_data(filtered_path, subsets_dir, news_data, out_root):
"""Split each subset into train/val; the test split is the news test set."""
with open(filtered_path) as f:
predictions = {d["articleId"]: d for d in map(json.loads, f)}
with open(news_data / "test.jsonl") as f:
news_test = f.readlines()
rng = random.Random(SELECTION["seed"]) # Share one RNG stream across subset sizes.
for key in SELECTION["subset_sizes"]:
with open(subsets_dir / f"subset_{key}.json") as f:
articles = [predictions[e["articleId"]] for e in json.load(f)]
rng.shuffle(articles)
n_val = max(1, int(len(articles) * SELECTION["validation_ratio"]))
out = out_root / key
out.mkdir(parents=True, exist_ok=True)
for name, part in (("train", articles[n_val:]), ("val", articles[:n_val])):
with open(out / f"{name}.jsonl", "w") as f:
for art in part:
f.write(json.dumps({"sentences": art["sentences"], "labels": art["predictions"],
"section_topic_labels": [], "sentence_topic_labels": []}) + "\n")
with open(out / "test.jsonl", "w") as f:
f.writelines(line if line.endswith("\n") else line + "\n" for line in news_test)
print(f"{key}: train={len(articles) - n_val}, val={n_val}, test={len(news_test)}")
|
main()
Source code in src/topic_segmentation/data/build_mind_silver_labels.py
111
112
113
114
115
116
117
118
119
120
121 | def main():
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("step", choices=["sentences", "subsets", "training-data"])
step = parser.parse_args().step
if step == "sentences":
write_sentences(MIND_DIR / "news.csv", REPOSITORY / CONFIG["sentence_file"])
elif step == "subsets":
write_subsets(REPOSITORY / CONFIG["teacher_predictions"], REPOSITORY / CONFIG["filtered_predictions"], MIND_DIR)
else:
write_training_data(REPOSITORY / CONFIG["filtered_predictions"], MIND_DIR, REPOSITORY / "data/news",
REPOSITORY / CONFIG["prepared_root"])
|