Skip to content

build_mind_silver_labels

topic_segmentation.data.build_mind_silver_labels

Prepare MIND sentences, silver labels, and training splits.

python -m topic_segmentation.data.build_mind_silver_labels sentences
python -m topic_segmentation.data.build_mind_silver_labels subsets
python -m topic_segmentation.data.build_mind_silver_labels training-data

scripts/prepare_data.py mind runs the news teacher between sentences and subsets. Settings come from configs/mind_adaptation.json. Subsets are prefixes of one shuffle; the test set is the IDEA-seg news test set.

clean_body(text)

Source code in src/topic_segmentation/data/build_mind_silver_labels.py
41
42
43
44
45
46
47
48
49
def clean_body(text):
    text = text.strip()
    for pattern in FOOTER_PATTERNS:
        match = pattern.search(text)
        if match and match.start() > len(text) * 0.5:
            text = text[:match.start()]
    text = re.sub(r' {2,}', ' ', text)
    text = re.sub(r' ([,\.!?;:\)])', r'\1', text)
    return text.strip()

write_sentences(input_path, output_path)

Clean bodies, split sentences, and keep articles within the sentence limits.

Source code in src/topic_segmentation/data/build_mind_silver_labels.py
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
def write_sentences(input_path, output_path):
    """Clean bodies, split sentences, and keep articles within the sentence limits."""
    with open(input_path, encoding="utf-8", newline="") as f:
        rows = list(csv.DictReader(f))
    kept = 0
    with open(output_path, "w", encoding="utf-8") as out:
        for row in rows:
            body = row.get("body", "")
            if not body:
                continue
            sentences = [s.strip() for s in sent_tokenize(clean_body(body)) if len(s.strip()) >= SELECTION["minimum_sentence_characters"]]
            if SELECTION["minimum_sentences"] <= len(sentences) <= SELECTION["maximum_sentences"]:
                out.write(json.dumps({"articleId": row["news_id"], "sentences": sentences}, ensure_ascii=False) + "\n")
                kept += 1
    print(f"kept {kept:,} of {len(rows):,} articles -> {output_path}")

write_subsets(predictions_path, filtered_path, subsets_dir)

Keep articles within the segment limits, shuffle once, and write nested index files.

Source code in src/topic_segmentation/data/build_mind_silver_labels.py
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
def write_subsets(predictions_path, filtered_path, subsets_dir):
    """Keep articles within the segment limits, shuffle once, and write nested index files."""
    records = []
    with open(predictions_path) as fin, open(filtered_path, "w") as fout:
        for line in fin:
            d = json.loads(line)
            if (SELECTION["minimum_segments"] <= sum(d["predictions"]) <= SELECTION["maximum_segments"]
                    and d["articleId"] not in EXCLUDED_ARTICLES):
                fout.write(line)
                records.append({"idx": len(records), "articleId": d["articleId"]})
    random.seed(SELECTION["seed"])
    random.shuffle(records)
    for name, size in SELECTION["subset_sizes"].items():
        with open(subsets_dir / f"subset_{name}.json", "w") as f:
            json.dump(records[:size], f)
    print(f"kept {len(records):,} articles -> {filtered_path}")

write_training_data(filtered_path, subsets_dir, news_data, out_root)

Split each subset into train/val; the test split is the news test set.

Source code in src/topic_segmentation/data/build_mind_silver_labels.py
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
def write_training_data(filtered_path, subsets_dir, news_data, out_root):
    """Split each subset into train/val; the test split is the news test set."""
    with open(filtered_path) as f:
        predictions = {d["articleId"]: d for d in map(json.loads, f)}
    with open(news_data / "test.jsonl") as f:
        news_test = f.readlines()
    rng = random.Random(SELECTION["seed"])  # Share one RNG stream across subset sizes.
    for key in SELECTION["subset_sizes"]:
        with open(subsets_dir / f"subset_{key}.json") as f:
            articles = [predictions[e["articleId"]] for e in json.load(f)]
        rng.shuffle(articles)
        n_val = max(1, int(len(articles) * SELECTION["validation_ratio"]))
        out = out_root / key
        out.mkdir(parents=True, exist_ok=True)
        for name, part in (("train", articles[n_val:]), ("val", articles[:n_val])):
            with open(out / f"{name}.jsonl", "w") as f:
                for art in part:
                    f.write(json.dumps({"sentences": art["sentences"], "labels": art["predictions"],
                                        "section_topic_labels": [], "sentence_topic_labels": []}) + "\n")
        with open(out / "test.jsonl", "w") as f:
            f.writelines(line if line.endswith("\n") else line + "\n" for line in news_test)
        print(f"{key}: train={len(articles) - n_val}, val={n_val}, test={len(news_test)}")

main()

Source code in src/topic_segmentation/data/build_mind_silver_labels.py
111
112
113
114
115
116
117
118
119
120
121
def main():
    parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
    parser.add_argument("step", choices=["sentences", "subsets", "training-data"])
    step = parser.parse_args().step
    if step == "sentences":
        write_sentences(MIND_DIR / "news.csv", REPOSITORY / CONFIG["sentence_file"])
    elif step == "subsets":
        write_subsets(REPOSITORY / CONFIG["teacher_predictions"], REPOSITORY / CONFIG["filtered_predictions"], MIND_DIR)
    else:
        write_training_data(REPOSITORY / CONFIG["filtered_predictions"], MIND_DIR, REPOSITORY / "data/news",
                            REPOSITORY / CONFIG["prepared_root"])