HTML to Plain Text Extractor (javascript, written by Codex)
envgap__codex__javascript-t1-34
Written by a coding agent; not on GitHubWritten 2026-03-03
01 / FAILURE SIGNATURE
As the study recorded it
None
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
package.json- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
codex/javascript-t1 #34 · read the task the agent was given
Codex wrote this javascript project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written. Task given to the agent: TASK: HTML to Plain Text Extractor Write a program that converts HTML documents to clean plain text, intelligently handling formatting, tables, lists, and links while removing all markup and scripts. FUNCTIONAL REQUIREMENTS: - Accept an HTML file path as a command-line argument - Strip all HTML tags, CSS styles, JavaScript, and comments while preserving readable text content - Convert HTML formatting to plain text equivalents: headings become UPPERCASE with underlines, bold text is wrapped in *asterisks*, lists become indented with bullets (- ) or numbers (1.), horizontal rules become dashed lines - Convert HTML tables to aligned plain text tables with column padding and separator rows - Convert hyperlinks to "text [URL]" format, or optionally strip URLs via --no-urls flag - Preserve paragraph spacing: consecutive block elements get blank line separators - Handle HTML entities: decode & < > — etc. to their text equivalents - Support extracting text from only specific HTML elements via --selector flag (CSS selector syntax, e.g., --selector "article" or --selector ".content") - Support extracting and listing all URLs found in the document via --extract-urls flag - Set maximum line width via --width flag (default: 80 characters) with word wrapping - Support batch conversion of multiple HTML files via --batch flag - Print the plain text output to console by default - Save to a file via --output flag (default: same base name with .txt extension) - If no input is given, generate a sample HTML page with headings, paragraphs, links, tables, lists, images, inline styles, scripts, and HTML entities, then convert it and display both the original HTML and the extracted text - Handle errors: malformed HTML (parse gracefully), encoding detection, and binary file detection Create a complete JavaScript project for a clean Ubuntu 22.04 machine with only Node.js 20+ (LTS) installed. Include: - Source code - package.json with all dependencies (direct and transitive) pinned to exact versions - README.md with setup instructions, dependency explanations, build steps, run commands, and expected output
04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
4 files, exactly as written, before any repair.
package-lock.json
{
"name": "html-to-text-extractor",
"version": "1.0.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "html-to-text-extractor",
"version": "1.0.0",
"license": "MIT"
}
}
}
package.json
{
"name": "html-to-text-extractor",
"version": "1.0.0",
"description": "Convert HTML files to clean plain text",
"type": "module",
"main": "src/index.js",
"scripts": {
"start": "node src/index.js"
},
"engines": {
"node": ">=20.0.0"
},
"license": "MIT"
}
README.md
# HTML to Plain Text Extractor (JavaScript) ## Requirements - Ubuntu 22.04 - Node.js 20+ ## Install ```bash npm install ``` No external runtime dependencies are used. The lockfile pins the dependency tree (empty in this project). ## Run ```bash node src/index.js input.html node src/index.js --no-urls --width 100 input.html node src/index.js --selector ".content" --extract-urls input.html node src/index.js --batch page1.html page2.html --output out_dir ``` If no input is provided, the tool generates `sample_page.html`, converts it, and prints both original and extracted output. ## Output - Console: extracted text table/list/link-aware rendering - Files: `.txt` output per input file - JSON summary: `html_extract_report.json` ## Features - Removes scripts/styles/comments - Decodes common HTML entities - Converts headings, lists, tables, links, rules to plain text equivalents - Supports selector extraction (`tag`, `.class`, `#id`) - URL extraction mode and batch conversion - Basic malformed/binary input handling
src/index.js
import fs from "node:fs";
import path from "node:path";
import process from "node:process";
function parseArgs(argv) {
const cfg = {
noUrls: false,
selector: null,
extractUrls: false,
width: 80,
batch: false,
output: null,
inputs: []
};
for (let i = 0; i < argv.length; i += 1) {
const a = argv[i];
if (!a.startsWith("--")) {
cfg.inputs.push(a);
continue;
}
if (a === "--no-urls") cfg.noUrls = true;
else if (a === "--extract-urls") cfg.extractUrls = true;
else if (a === "--batch") cfg.batch = true;
else if (a === "--selector") cfg.selector = argv[++i];
else if (a === "--width") cfg.width = Number.parseInt(argv[++i], 10);
else if (a === "--output") cfg.output = argv[++i];
else throw new Error(`Unknown option: ${a}`);
}
if (!Number.isInteger(cfg.width) || cfg.width < 20) throw new Error("--width must be an integer >= 20");
return cfg;
}
function isBinary(buffer) {
const n = Math.min(buffer.length, 1024);
for (let i = 0; i < n; i += 1) if (buffer[i] === 0) return true;
return false;
}
function decodeEntities(text) {
const named = {
nbsp: " ",
amp: "&",
lt: "<",
gt: ">",
quot: '"',
apos: "'",
mdash: "—",
ndash: "–",
copy: "©",
reg: "®",
trade: "™",
hellip: "…"
};
return text.replace(/&(#x[0-9a-fA-F]+|#\d+|[a-zA-Z]+);/g, (_, code) => {
if (code.startsWith("#x")) {
const num = Number.parseInt(code.slice(2), 16);
return Number.isNaN(num) ? _ : String.fromCodePoint(num);
}
if (code.startsWith("#")) {
const num = Number.parseInt(code.slice(1), 10);
return Number.isNaN(num) ? _ : String.fromCodePoint(num);
}
return named[code] ?? _;
});
}
function stripScriptsStylesComments(html) {
return html
.replace(/<!--[\s\S]*?-->/g, "")
.replace(/<script\b[^>]*>[\s\S]*?<\/script>/gi, "")
.replace(/<style\b[^>]*>[\s\S]*?<\/style>/gi, "");
}
function extractUrls(html) {
const urls = new Set();
const rx = /\b(?:href|src)\s*=\s*["']([^"']+)["']/gi;
let m;
while ((m = rx.exec(html)) !== null) urls.add(m[1]);
return [...urls];
}
function selectHtml(html, selector) {
if (!selector) return html;
const parts = selector.trim().split(/\s*,\s*/);
const chunks = [];
for (const part of parts) {
if (part.startsWith(".")) {
const cls = part.slice(1).replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
const rx = new RegExp(`<([a-zA-Z0-9]+)([^>]*\\bclass=["'][^"']*\\b${cls}\\b[^"']*["'][^>]*)>([\\s\\S]*?)<\\/\\1>`, "gi");
let m;
while ((m = rx.exec(html)) !== null) chunks.push(m[0]);
} else if (part.startsWith("#")) {
const id = part.slice(1).replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
const rx = new RegExp(`<([a-zA-Z0-9]+)([^>]*\\bid=["']${id}["'][^>]*)>([\\s\\S]*?)<\\/\\1>`, "gi");
let m;
while ((m = rx.exec(html)) !== null) chunks.push(m[0]);
} else {
const tag = part.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
const rx = new RegExp(`<${tag}\\b[^>]*>([\\s\\S]*?)<\\/${tag}>`, "gi");
let m;
while ((m = rx.exec(html)) !== null) chunks.push(m[0]);
}
}
return chunks.length ? chunks.join("\n") : "";
}
function wrapLines(text, width) {
const out = [];
for (const rawLine of text.split("\n")) {
const line = rawLine.trimEnd();
if (!line.trim()) {
out.push("");
continue;
}
if (line.length <= width) {
out.push(line);
continue;
}
let cur = line;
while (cur.length > width) {
let cut = cur.lastIndexOf(" ", width);
if (cut < 10) cut = width;
out.push(cur.slice(0, cut));
cur = cur.slice(cut).trimStart();
}
if (cur) out.push(cur);
}
return out.join("\n");
}
function renderTable(tableHtml) {
const rows = [];
const rowRx = /<tr\b[^>]*>([\s\S]*?)<\/tr>/gi;
let rm;
while ((rm = rowRx.exec(tableHtml)) !== null) {
const cells = [];
const cellRx = /<(th|td)\b[^>]*>([\s\S]*?)<\/\1>/gi;
let cm;
while ((cm = cellRx.exec(rm[1])) !== null) {
const txt = decodeEntities(cm[2].replace(/<[^>]+>/g, " ").replace(/\s+/g, " ").trim());
cells.push(txt);
}
if (cells.length) rows.push(cells);
}
if (!rows.length) return "";
const cols = Math.max(...rows.map((r) => r.length));
const widths = new Array(cols).fill(0);
for (const r of rows) {
for (let i = 0; i < cols; i += 1) {
const val = r[i] ?? "";
widths[i] = Math.max(widths[i], val.length);
}
}
const lines = [];
for (let ri = 0; ri < rows.length; ri += 1) {
const r = rows[ri];
const line = [];
for (let i = 0; i < cols; i += 1) {
const v = (r[i] ?? "").padEnd(widths[i], " ");
line.push(v);
}
lines.push(`| ${line.join(" | ")} |`);
if (ri === 0) {
lines.push(`| ${widths.map((w) => "-".repeat(Math.max(3, w))).join(" | ")} |`);
}
}
return `\n${lines.join("\n")}\n`;
}
function renderList(listHtml, ordered) {
const liRx = /<li\b[^>]*>([\s\S]*?)<\/li>/gi;
const lines = [];
let i = 1;
let m;
while ((m = liRx.exec(listHtml)) !== null) {
const txt = decodeEntities(m[1].replace(/<[^>]+>/g, " ").replace(/\s+/g, " ").trim());
lines.push(`${ordered ? `${i}.` : "-"} ${txt}`);
i += 1;
}
return `\n${lines.join("\n")}\n`;
}
function convertHtmlToText(inputHtml, cfg) {
let html = stripScriptsStylesComments(inputHtml);
const urls = extractUrls(html);
html = selectHtml(html, cfg.selector);
html = html.replace(/<table\b[^>]*>[\s\S]*?<\/table>/gi, (m) => renderTable(m));
html = html.replace(/<ol\b[^>]*>[\s\S]*?<\/ol>/gi, (m) => renderList(m, true));
html = html.replace(/<ul\b[^>]*>[\s\S]*?<\/ul>/gi, (m) => renderList(m, false));
html = html.replace(/<h([1-6])\b[^>]*>([\s\S]*?)<\/h\1>/gi, (_, level, content) => {
const txt = decodeEntities(content.replace(/<[^>]+>/g, " ").replace(/\s+/g, " ").trim()).toUpperCase();
const under = (level <= 2 ? "=" : "-").repeat(Math.max(3, txt.length));
return `\n${txt}\n${under}\n`;
});
html = html.replace(/<(strong|b)\b[^>]*>([\s\S]*?)<\/\1>/gi, (_, _tag, content) => `*${decodeEntities(content.replace(/<[^>]+>/g, " ").trim())}*`);
html = html.replace(/<hr\b[^>]*\/?>/gi, "\n" + "-".repeat(40) + "\n");
html = html.replace(/<a\b[^>]*href=["']([^"']+)["'][^>]*>([\s\S]*?)<\/a>/gi, (_, href, text) => {
const t = decodeEntities(text.replace(/<[^>]+>/g, " ").replace(/\s+/g, " ").trim());
return cfg.noUrls ? t : `${t} [${href}]`;
});
html = html
.replace(/<br\b[^>]*\/?>/gi, "\n")
.replace(/<\/?(p|div|section|article|header|footer|main|aside|nav|figure|figcaption|blockquote)\b[^>]*>/gi, "\n\n")
.replace(/<\/?(tr)\b[^>]*>/gi, "\n")
.replace(/<\/?(td|th)\b[^>]*>/gi, " ");
let text = html.replace(/<[^>]+>/g, " ");
text = decodeEntities(text);
text = text
.replace(/[ \t]+\n/g, "\n")
.replace(/\n[ \t]+/g, "\n")
.replace(/\n{3,}/g, "\n\n")
.replace(/[ \t]{2,}/g, " ")
.trim();
text = wrapLines(text, cfg.width);
return { text, urls };
}
function createSampleHtml() {
const sample = `<!doctype html>
<html>
<head>
<title>Sample HTML</title>
<style>.red{color:red}</style>
<script>console.log("ignore me")</script>
</head>
<body>
<!-- comment to remove -->
<article class="content" id="main">
<h1>Demo Article</h1>
<p>This is a <strong>sample</strong> paragraph with an <a href="https://example.com">example link</a> & entities like — and spaces.</p>
<h2>List Section</h2>
<ul><li>First bullet</li><li>Second bullet</li></ul>
<ol><li>First numbered</li><li>Second numbered</li></ol>
<hr />
<h2>Table Section</h2>
<table>
<tr><th>Name</th><th>Role</th><th>Score</th></tr>
<tr><td>Ada</td><td>Engineer</td><td>98</td></tr>
<tr><td>Linus</td><td>Maintainer</td><td>95</td></tr>
</table>
<p>Image source should be captured: <img src="https://cdn.example.com/img.png" alt="img"></p>
</article>
</body>
</html>`;
const p = path.resolve("sample_page.html");
fs.writeFileSync(p, sample, "utf8");
return p;
}
function outputPathForInput(inputFile, cfg, batchMode) {
if (cfg.output && !batchMode) return path.resolve(cfg.output);
if (cfg.output && batchMode) return path.resolve(cfg.output);
const parsed = path.parse(inputFile);
return path.join(parsed.dir, `${parsed.name}.txt`);
}
function runOne(inputFile, cfg, batchMode = false) {
const buffer = fs.readFileSync(inputFile);
if (isBinary(buffer)) throw new Error(`Binary file detected: ${inputFile}`);
let html = buffer.toString("utf8");
const { text, urls } = convertHtmlToText(html, cfg);
const outPath = outputPathForInput(inputFile, cfg, batchMode);
if (!batchMode) {
fs.writeFileSync(outPath, `${text}\n`, "utf8");
}
process.stdout.write(`${text}\n`);
if (cfg.extractUrls) {
process.stdout.write(`\nURLs:\n${urls.map((u) => `- ${u}`).join("\n")}\n`);
}
return { inputFile, outPath, text, urls };
}
function main() {
try {
const cfg = parseArgs(process.argv.slice(2));
if (!cfg.inputs.length) {
const samplePath = createSampleHtml();
const html = fs.readFileSync(samplePath, "utf8");
const { text, urls } = convertHtmlToText(html, cfg);
const outPath = path.resolve("sample_page.txt");
fs.writeFileSync(outPath, `${text}\n`, "utf8");
process.stdout.write("===== ORIGINAL HTML =====\n");
process.stdout.write(`${html}\n`);
process.stdout.write("\n===== EXTRACTED TEXT =====\n");
process.stdout.write(`${text}\n`);
if (cfg.extractUrls) process.stdout.write(`\nURLs:\n${urls.map((u) => `- ${u}`).join("\n")}\n`);
return;
}
const targets = cfg.batch ? cfg.inputs : [cfg.inputs[0]];
const all = [];
if (cfg.batch && cfg.output) fs.mkdirSync(path.resolve(cfg.output), { recursive: true });
for (const input of targets) {
if (!fs.existsSync(input)) {
process.stderr.write(`Warning: missing file ${input}\n`);
continue;
}
try {
if (cfg.batch) {
const buffer = fs.readFileSync(input);
if (isBinary(buffer)) throw new Error(`Binary file detected: ${input}`);
const html = buffer.toString("utf8");
const { text, urls } = convertHtmlToText(html, cfg);
const outDir = cfg.output ? path.resolve(cfg.output) : path.dirname(path.resolve(input));
const outPath = path.join(outDir, `${path.parse(input).name}.txt`);
fs.writeFileSync(outPath, `${text}\n`, "utf8");
process.stdout.write(`${text}\n`);
if (cfg.extractUrls) process.stdout.write(`\nURLs for ${input}:\n${urls.map((u) => `- ${u}`).join("\n")}\n`);
all.push({ input, outPath, urlsCount: urls.length });
} else {
const r = runOne(input, cfg, false);
all.push({ input: r.inputFile, outPath: r.outPath, urlsCount: r.urls.length });
}
} catch (err) {
process.stderr.write(`Warning: ${err.message}\n`);
}
}
const report = { generatedAt: new Date().toISOString(), files: all };
fs.writeFileSync("html_extract_report.json", JSON.stringify(report, null, 2), "utf8");
} catch (error) {
process.stderr.write(`Error: ${error.message}\n`);
process.exit(1);
}
}
main();