Skip to content

build_subheadings

topic_segmentation.news_dataset.build_subheadings

Select saved articles with at least two content subheadings.

Copy matching pages from data/pages/raw_html/ to data/pages/<outlet>_subheadings/, using outlet-specific HTML selectors and promo filters.

python -m topic_segmentation.news_dataset.build_subheadings BBC
python -m topic_segmentation.news_dataset.build_subheadings all

is_noise(text: str, noise_re: re.Pattern) -> bool

Source code in src/topic_segmentation/news_dataset/build_subheadings.py
48
49
50
def is_noise(text: str, noise_re: re.Pattern) -> bool:
    t = text.strip()
    return len(t) < 3 or bool(noise_re.search(t))

has_subheadings_default(soup) -> bool

Find at least two content h2/h3 headings inside article.

Source code in src/topic_segmentation/news_dataset/build_subheadings.py
55
56
57
58
59
60
61
62
63
64
def has_subheadings_default(soup) -> bool:
    """Find at least two content h2/h3 headings inside article."""
    article = soup.select_one("article")
    if not article:
        return False
    real = sum(
        1 for h in article.find_all(["h2", "h3"])
        if not is_noise(h.get_text(strip=True), COMMON_RE)
    )
    return real >= 2

has_subheadings_es(soup) -> bool

Find Evening Standard headings in #main, excluding empty Opta articles.

Source code in src/topic_segmentation/news_dataset/build_subheadings.py
67
68
69
70
71
72
73
74
75
76
77
78
79
def has_subheadings_es(soup) -> bool:
    """Find Evening Standard headings in #main, excluding empty Opta articles."""
    widgets = soup.find_all("opta-widget")
    if widgets and all(len(w.get_text(strip=True)) == 0 for w in widgets):
        return False
    main = soup.find(id="main")
    if not main:
        return False
    real = sum(
        1 for h in main.find_all(["h2", "h3", "h4"])
        if not is_noise(h.get_text(strip=True), ES_RE)
    )
    return real >= 2

has_subheadings_independent(soup) -> bool

Find Independent section headings in div.sc-kk992l-0.

Source code in src/topic_segmentation/news_dataset/build_subheadings.py
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
def has_subheadings_independent(soup) -> bool:
    """Find Independent section headings in div.sc-kk992l-0."""
    article = soup.find("article")
    if not article:
        return False
    main_wrapper = article.find("div", class_=lambda c: c and "main-wrapper" in c)
    if not main_wrapper:
        return False
    content = main_wrapper.find("div", class_=lambda c: c and "sc-jbiisr-0" in c)
    if not content:
        return False
    real = sum(
        1 for d in content.find_all("div", class_=lambda c: c and "sc-kk992l-0" in c)
        if d.find(["h2", "h3", "h4"])
        and not is_noise(d.get_text(strip=True), COMMON_RE)
        and len(d.get_text(strip=True)) <= 200
    )
    return real >= 2

has_subheadings_inews(soup) -> bool

Find iNews headings in div.article-content, excluding Read Next widgets.

Source code in src/topic_segmentation/news_dataset/build_subheadings.py
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
def has_subheadings_inews(soup) -> bool:
    """Find iNews headings in div.article-content, excluding Read Next widgets."""
    content = soup.find("div", class_="article-content")
    if not content:
        return False
    real = 0
    for h in content.find_all(["h2", "h3"]):
        if h.find_parent(class_="inews__shortcode-relatedarticleinline__content"):
            continue
        if h.find_parent(class_="inews__shortcode-relatedarticleinline"):
            continue
        if is_noise(h.get_text(strip=True), COMMON_RE):
            continue
        real += 1
    return real >= 2

is_sky_news_strong_heading(el) -> bool

Recognize a <p><strong> heading of at most 100 characters.

Source code in src/topic_segmentation/news_dataset/build_subheadings.py
126
127
128
129
130
131
132
133
134
135
136
137
138
139
def is_sky_news_strong_heading(el) -> bool:
    """Recognize a `<p><strong>` heading of at most 100 characters."""
    if el.name != "p":
        return False
    direct_text = "".join(str(c) for c in el.children if c.name is None).strip()
    if direct_text:
        return False
    children = [c for c in el.children if c.name is not None]
    if len(children) != 1 or children[0].name != "strong":
        return False
    txt = el.get_text(" ", strip=True)
    if len(txt) > 100:
        return False
    return True

has_subheadings_sky_news(soup) -> bool

Find Sky News bold paragraph headings in div.sdc-article-body.

Source code in src/topic_segmentation/news_dataset/build_subheadings.py
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
def has_subheadings_sky_news(soup) -> bool:
    """Find Sky News bold paragraph headings in div.sdc-article-body."""
    body = soup.find("div", class_="sdc-article-body")
    if not body:
        return False
    real = 0
    for p in body.find_all("p"):
        if not is_sky_news_strong_heading(p):
            continue
        txt = p.get_text(" ", strip=True)
        tl = txt.lower()
        if any(tl.startswith(n) for n in SKY_NEWS_NOISE_PREFIXES):
            continue
        if tl in SKY_NEWS_NOISE_FULL:
            continue
        real += 1
    return real >= 2

has_subheadings(html: str, outlet: str) -> bool

Source code in src/topic_segmentation/news_dataset/build_subheadings.py
161
162
163
164
165
166
167
168
169
170
171
def has_subheadings(html: str, outlet: str) -> bool:
    soup = BeautifulSoup(ftfy.fix_text(html), "html.parser")
    if outlet == "EveningStandard":
        return has_subheadings_es(soup)
    if outlet == "Independent":
        return has_subheadings_independent(soup)
    if outlet == "INews":
        return has_subheadings_inews(soup)
    if outlet == "SkyNews":
        return has_subheadings_sky_news(soup)
    return has_subheadings_default(soup)

inspect_article(aid, outlet)

Source code in src/topic_segmentation/news_dataset/build_subheadings.py
187
188
189
190
191
def inspect_article(aid, outlet):
    source = RAW_HTML_DIR / f"{aid}.html"
    if not source.exists():
        return aid, False
    return aid, has_subheadings(source.read_text(encoding="utf-8"), outlet)

build_outlet(outlet: str, articles: list) -> None

Source code in src/topic_segmentation/news_dataset/build_subheadings.py
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
def build_outlet(outlet: str, articles: list) -> None:
    dir_name = OUTLET_DIR_NAMES.get(outlet, f"{outlet.lower()}_subheadings")
    output_dir = ROOT / "data" / "pages" / dir_name

    if output_dir.exists():
        shutil.rmtree(output_dir)
    output_dir.mkdir(parents=True)

    # Deduplicate by URL, keep earliest articleId
    seen_urls: set = set()
    ids: set = set()
    for a in sorted(articles, key=lambda x: x["articleId"]):
        if a.get("outlet") != outlet:
            continue
        url = a.get("url", "")
        if url and url not in seen_urls:
            seen_urls.add(url)
            ids.add(a["articleId"])
    print(f"{outlet} articles (deduplicated): {len(ids)}", flush=True)

    found = 0
    workers = max(1, int(os.environ.get("TS_NEWS_WORKERS", "1")))
    with ProcessPoolExecutor(max_workers=workers) as executor:
        for aid, keep in executor.map(partial(inspect_article, outlet=outlet), sorted(ids)):
            if keep:
                shutil.copy2(RAW_HTML_DIR / f"{aid}.html", output_dir / f"{aid}.html")
                found += 1

    print(f"Copied {found} articles with subheadings -> {output_dir}/", flush=True)

main()

Source code in src/topic_segmentation/news_dataset/build_subheadings.py
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
def main():
    parser = argparse.ArgumentParser(description="Build <outlet>_subheadings/ directory")
    parser.add_argument(
        "outlet",
        help="Outlet name as in articles.json (e.g. BBC, EveningStandard, GuardianInt, "
             "INews, Independent, SkyNews) — or 'all' to run every outlet",
    )
    args = parser.parse_args()

    with open(ARTICLES_JSON, encoding="latin-1") as f:
        articles = json.load(f)

    outlets = list(OUTLET_DIR_NAMES) if args.outlet == "all" else [args.outlet]
    for i, outlet in enumerate(outlets):
        if i > 0:
            print()
        build_outlet(outlet, articles)