Repository object · research-note

Normalize Html Visible Text

Accepted research note in the public catalog.

Source path
research/how-we-know/gpt-4-bar-exam-percentile/normalize_html_visible_text.py
Media type
text/x-python
Object ID
em:research-note:sha256:f0bcc1feb7a9e4baad6fae321ebdcce35565d73acb750cfbee2e79cfae6a9efd
Content digest
379737bf5ad58278eee0ee09d2e3c2c621fc022041333bd264a53ec8f0bfb575

Source content

"""Deterministically extract disclosure-safe visible text from captured HTML bytes."""

from __future__ import annotations

import argparse

import re

import sys

from html.parser import HTMLParser

from pathlib import Path

EXCLUDED_ELEMENTS = frozenset({"script", "style", "template", "noscript", "svg"})

class VisibleTextExtractor(HTMLParser):

"""Collect visible text under one exact HTML `id` attribute."""

def __init__(self, root_id: str) -> None:

super().__init__(convert_charrefs=True)

self.root_id = root_id

self.parts: list[str] = []

self.root_tag: str | None = None

self.root_nesting = 0

self.root_matches = 0

self.excluded_depth = 0

def handle_starttag(

self,

tag: str,

attrs: list[tuple[str, str | None]],

) -> None:

attributes = dict(attrs)

if attributes.get("id") == self.root_id:

self.root_matches += 1

if self.root_tag is None:

self.root_tag = tag.lower()

self.root_nesting = 1

elif self.root_tag is not None and tag.lower() == self.root_tag:

self.root_nesting += 1

if self.root_tag is not None and tag.lower() in EXCLUDED_ELEMENTS:

self.excluded_depth += 1

def handle_endtag(self, tag: str) -> None:

if self.root_tag is not None and tag.lower() in EXCLUDED_ELEMENTS:

if not self.excluded_depth:

raise ValueError(f"unbalanced excluded element: {tag}")

self.excluded_depth -= 1

if self.root_tag is not None and tag.lower() == self.root_tag:

self.root_nesting -= 1

if not self.root_nesting:

self.root_tag = None

def handle_startendtag(

self,

tag: str,

attrs: list[tuple[str, str | None]],

) -> None:

self.handle_starttag(tag, attrs)

self.handle_endtag(tag)

def handle_data(self, data: str) -> None:

if self.root_tag is not None and not self.excluded_depth:

self.parts.append(data)

def normalize_html_visible_text(payload: bytes, *, root_id: str) -> bytes:

"""Return UTF-8 text with excluded elements removed and whitespace collapsed."""

text = payload.decode("utf-8")

parser = VisibleTextExtractor(root_id)

parser.feed(text)

parser.close()

if parser.root_tag is not None:

raise ValueError(f"unclosed selected element: {parser.root_tag}")

if parser.root_matches != 1:

raise ValueError(

f"expected one element with id={root_id!r}; found {parser.root_matches}"

)

normalized = re.sub(r"\s+", " ", " ".join(parser.parts)).strip()

return normalized.encode("utf-8")

def main() -> int:

parser = argparse.ArgumentParser()

parser.add_argument("--collapse-whitespace", action="store_true", required=True)

parser.add_argument("--root-id", required=True)

parser.add_argument("input", type=Path, nargs="?")

args = parser.parse_args()

payload = args.input.read_bytes() if args.input else sys.stdin.buffer.read()

sys.stdout.buffer.write(normalize_html_visible_text(payload, root_id=args.root_id))

return 0

if __name__ == "__main__":

raise SystemExit(main())

Build receipt

Reproduce this projection

Reproducible projection
Catalog
em:catalog:sha256:9bfc972213cba2cde167386103dc2c011ee74639fb7f0794c54120fbbdef1a5d
Frontier
em:frontier:sha256:f33be3eae4c75232d56750ef9a1aa79d96274ece3417d65a75c1391bf61a81bf
Accepted commit
f92846570180dfa4511263f8ba98ecd18f7772c9
Epistemic policy
commons-balanced-v0.1
Disclosure policy
public-noninterference-v0.1
Compiler
epistemedia/0.2.0