HTML to Plain Text Extractor (python, written by Codex)
envgap__codex__python-t1-34
Written by a coding agent; not on GitHubWritten 2026-03-03
01 / FAILURE SIGNATURE
As the study recorded it
None
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
requirements.txt- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
codex/python-t1 #34 · read the task the agent was given
Codex wrote this python project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written. Task given to the agent: TASK: HTML to Plain Text Extractor Write a program that converts HTML documents to clean plain text, intelligently handling formatting, tables, lists, and links while removing all markup and scripts. FUNCTIONAL REQUIREMENTS: - Accept an HTML file path as a command-line argument - Strip all HTML tags, CSS styles, JavaScript, and comments while preserving readable text content - Convert HTML formatting to plain text equivalents: headings become UPPERCASE with underlines, bold text is wrapped in *asterisks*, lists become indented with bullets (- ) or numbers (1.), horizontal rules become dashed lines - Convert HTML tables to aligned plain text tables with column padding and separator rows - Convert hyperlinks to "text [URL]" format, or optionally strip URLs via --no-urls flag - Preserve paragraph spacing: consecutive block elements get blank line separators - Handle HTML entities: decode & < > — etc. to their text equivalents - Support extracting text from only specific HTML elements via --selector flag (CSS selector syntax, e.g., --selector "article" or --selector ".content") - Support extracting and listing all URLs found in the document via --extract-urls flag - Set maximum line width via --width flag (default: 80 characters) with word wrapping - Support batch conversion of multiple HTML files via --batch flag - Print the plain text output to console by default - Save to a file via --output flag (default: same base name with .txt extension) - If no input is given, generate a sample HTML page with headings, paragraphs, links, tables, lists, images, inline styles, scripts, and HTML entities, then convert it and display both the original HTML and the extracted text - Handle errors: malformed HTML (parse gracefully), encoding detection, and binary file detection Create a complete Python project for a clean Ubuntu 22.04 machine with only Python 3.10+ installed. Include: - Source code - requirements.txt with all dependencies (direct and transitive) pinned to exact versions - README.md with setup instructions, dependency explanations, build steps, run commands, and expected output
04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
3 files, exactly as written, before any repair.
README.md
# HTML to Plain Text Extractor (Python) ## Requirements - Ubuntu 22.04 - Python 3.10+ ## Install ```bash python3 -m venv .venv source .venv/bin/activate pip install -r requirements.txt ``` No third-party packages are required. ## Run ```bash python src/main.py input.html python src/main.py --no-urls --extract-urls --width 100 input.html python src/main.py --selector ".content" --batch a.html b.html --output out_dir ``` If no inputs are provided, a sample HTML page is generated and analyzed. ## Output - Extracted text printed to console - `.txt` file written per input - JSON operation report: `html_extract_report.json`
requirements.txt
# No external dependencies required
src/main.py
#!/usr/bin/env python3
import argparse
import html
import json
import os
import re
import sys
import textwrap
from pathlib import Path
def is_binary(path: Path) -> bool:
data = path.read_bytes()[:1024]
return b"\x00" in data
def strip_noise(raw: str) -> str:
raw = re.sub(r"<!--.*?-->", "", raw, flags=re.S)
raw = re.sub(r"<script\b[^>]*>.*?</script>", "", raw, flags=re.S | re.I)
raw = re.sub(r"<style\b[^>]*>.*?</style>", "", raw, flags=re.S | re.I)
return raw
def extract_urls(raw: str) -> list[str]:
urls = re.findall(r"\b(?:href|src)\s*=\s*[\"']([^\"']+)[\"']", raw, flags=re.I)
return sorted(set(urls))
def select_html(raw: str, selector: str | None) -> str:
if not selector:
return raw
chunks: list[str] = []
for part in [x.strip() for x in selector.split(",") if x.strip()]:
if part.startswith("."):
cls = re.escape(part[1:])
pattern = rf"<([a-zA-Z0-9]+)([^>]*\bclass=[\"'][^\"']*\b{cls}\b[^\"']*[\"'][^>]*)>(.*?)</\1>"
elif part.startswith("#"):
sid = re.escape(part[1:])
pattern = rf"<([a-zA-Z0-9]+)([^>]*\bid=[\"']{sid}[\"'][^>]*)>(.*?)</\1>"
else:
tag = re.escape(part)
pattern = rf"<{tag}\b[^>]*>(.*?)</{tag}>"
chunks.extend(m.group(0) for m in re.finditer(pattern, raw, flags=re.S | re.I))
return "\n".join(chunks)
def render_table(block: str) -> str:
rows = []
for row in re.finditer(r"<tr\b[^>]*>(.*?)</tr>", block, flags=re.S | re.I):
vals = []
for c in re.finditer(r"<(?:th|td)\b[^>]*>(.*?)</(?:th|td)>", row.group(1), flags=re.S | re.I):
txt = html.unescape(re.sub(r"<[^>]+>", " ", c.group(1)))
vals.append(re.sub(r"\s+", " ", txt).strip())
if vals:
rows.append(vals)
if not rows:
return ""
cols = max(len(r) for r in rows)
widths = [0] * cols
for r in rows:
for i in range(cols):
widths[i] = max(widths[i], len(r[i]) if i < len(r) else 0)
lines = []
for i, r in enumerate(rows):
line = " | ".join((r[j] if j < len(r) else "").ljust(widths[j]) for j in range(cols))
lines.append(f"| {line} |")
if i == 0:
sep = " | ".join("-" * max(3, w) for w in widths)
lines.append(f"| {sep} |")
return "\n" + "\n".join(lines) + "\n"
def render_list(block: str, ordered: bool) -> str:
items = []
for i, m in enumerate(re.finditer(r"<li\b[^>]*>(.*?)</li>", block, flags=re.S | re.I), start=1):
txt = html.unescape(re.sub(r"<[^>]+>", " ", m.group(1)))
txt = re.sub(r"\s+", " ", txt).strip()
items.append(f"{i}. {txt}" if ordered else f"- {txt}")
return "\n" + "\n".join(items) + "\n"
def wrap_text(txt: str, width: int) -> str:
out = []
for line in txt.splitlines():
if not line.strip():
out.append("")
continue
out.extend(textwrap.wrap(line, width=width, break_long_words=False) or [line])
return "\n".join(out)
def convert(raw_html: str, selector: str | None, no_urls: bool, width: int) -> tuple[str, list[str]]:
raw_html = strip_noise(raw_html)
urls = extract_urls(raw_html)
raw_html = select_html(raw_html, selector)
raw_html = re.sub(r"<table\b[^>]*>.*?</table>", lambda m: render_table(m.group(0)), raw_html, flags=re.S | re.I)
raw_html = re.sub(r"<ol\b[^>]*>.*?</ol>", lambda m: render_list(m.group(0), True), raw_html, flags=re.S | re.I)
raw_html = re.sub(r"<ul\b[^>]*>.*?</ul>", lambda m: render_list(m.group(0), False), raw_html, flags=re.S | re.I)
def hconv(m: re.Match) -> str:
text = html.unescape(re.sub(r"<[^>]+>", " ", m.group(2)))
text = re.sub(r"\s+", " ", text).strip().upper()
line = "=" * max(3, len(text)) if int(m.group(1)) <= 2 else "-" * max(3, len(text))
return f"\n{text}\n{line}\n"
raw_html = re.sub(r"<h([1-6])\b[^>]*>(.*?)</h\1>", hconv, raw_html, flags=re.S | re.I)
raw_html = re.sub(r"<(?:strong|b)\b[^>]*>(.*?)</(?:strong|b)>", lambda m: f"*{html.unescape(re.sub(r'<[^>]+>', ' ', m.group(1))).strip()}*", raw_html, flags=re.S | re.I)
raw_html = re.sub(r"<hr\b[^>]*?/?>", "\n" + "-" * 40 + "\n", raw_html, flags=re.I)
def link_conv(m: re.Match) -> str:
href, txt = m.group(1), html.unescape(re.sub(r"<[^>]+>", " ", m.group(2)))
txt = re.sub(r"\s+", " ", txt).strip()
return txt if no_urls else f"{txt} [{href}]"
raw_html = re.sub(r"<a\b[^>]*href=[\"']([^\"']+)[\"'][^>]*>(.*?)</a>", link_conv, raw_html, flags=re.S | re.I)
raw_html = re.sub(r"<br\b[^>]*?/?>", "\n", raw_html, flags=re.I)
raw_html = re.sub(r"</?(?:p|div|section|article|header|footer|main|aside|nav|figure|figcaption|blockquote)\b[^>]*>", "\n\n", raw_html, flags=re.I)
txt = re.sub(r"<[^>]+>", " ", raw_html)
txt = html.unescape(txt)
txt = re.sub(r"[ \t]+\n", "\n", txt)
txt = re.sub(r"\n{3,}", "\n\n", txt)
txt = re.sub(r"[ \t]{2,}", " ", txt).strip()
return wrap_text(txt, width), urls
def create_sample(path: Path) -> None:
sample = """<!doctype html>
<html><head><title>Sample</title><style>.a{color:red}</style><script>console.log('x')</script></head>
<body>
<article class='content' id='main'>
<h1>Demo Article</h1>
<p>This is a <b>sample</b> paragraph with <a href='https://example.com'>link</a> & — entities.</p>
<h2>List</h2>
<ul><li>Alpha</li><li>Beta</li></ul>
<ol><li>One</li><li>Two</li></ol>
<hr>
<table><tr><th>Name</th><th>Score</th></tr><tr><td>Ada</td><td>98</td></tr><tr><td>Linus</td><td>95</td></tr></table>
<img src='https://cdn.example.com/image.png'>
</article>
</body></html>"""
path.write_text(sample, encoding="utf-8")
def main() -> int:
parser = argparse.ArgumentParser(description="HTML to plain text extractor")
parser.add_argument("inputs", nargs="*", help="HTML input files")
parser.add_argument("--no-urls", action="store_true")
parser.add_argument("--selector")
parser.add_argument("--extract-urls", action="store_true")
parser.add_argument("--width", type=int, default=80)
parser.add_argument("--batch", action="store_true")
parser.add_argument("--output", help="Output file or output dir in batch mode")
args = parser.parse_args()
if args.width < 20:
print("Error: --width must be >= 20", file=sys.stderr)
return 1
if not args.inputs:
sample = Path("sample_page.html")
create_sample(sample)
raw = sample.read_text(encoding="utf-8", errors="replace")
text, urls = convert(raw, args.selector, args.no_urls, args.width)
Path("sample_page.txt").write_text(text + "\n", encoding="utf-8")
print("===== ORIGINAL HTML =====")
print(raw)
print("\n===== EXTRACTED TEXT =====")
print(text)
if args.extract_urls:
print("\nURLs:")
for u in urls:
print(f"- {u}")
return 0
targets = args.inputs if args.batch else args.inputs[:1]
report = {"generatedAt": __import__("datetime").datetime.utcnow().isoformat() + "Z", "files": []}
out_dir = Path(args.output).resolve() if args.batch and args.output else None
if out_dir:
out_dir.mkdir(parents=True, exist_ok=True)
for p in targets:
path = Path(p)
if not path.exists():
print(f"Warning: missing file {p}", file=sys.stderr)
continue
try:
if is_binary(path):
raise ValueError("Binary file detected")
raw = path.read_text(encoding="utf-8", errors="replace")
text, urls = convert(raw, args.selector, args.no_urls, args.width)
if args.batch:
out_path = (out_dir if out_dir else path.parent) / f"{path.stem}.txt"
else:
out_path = Path(args.output).resolve() if args.output else path.with_suffix(".txt")
out_path.write_text(text + "\n", encoding="utf-8")
print(text)
if args.extract_urls:
print(f"\nURLs for {path}:")
for u in urls:
print(f"- {u}")
report["files"].append({"input": str(path), "output": str(out_path), "urlsCount": len(urls)})
except Exception as exc:
print(f"Warning: {path}: {exc}", file=sys.stderr)
Path("html_extract_report.json").write_text(json.dumps(report, indent=2), encoding="utf-8")
return 0
if __name__ == "__main__":
raise SystemExit(main())