HTML to Plain Text Extractor (cpp, written by Codex)
envgap__codex__cpp-t1-34
Written by a coding agent; not on GitHubWritten 2026-03-03
01 / FAILURE SIGNATURE
As the study recorded it
No identifying execution failure has been captured.
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
CMakeLists.txt- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
codex/cpp-t1 #34 · read the task the agent was given
Codex wrote this cpp project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written. Task given to the agent: TASK: HTML to Plain Text Extractor Write a program that converts HTML documents to clean plain text, intelligently handling formatting, tables, lists, and links while removing all markup and scripts. FUNCTIONAL REQUIREMENTS: - Accept an HTML file path as a command-line argument - Strip all HTML tags, CSS styles, JavaScript, and comments while preserving readable text content - Convert HTML formatting to plain text equivalents: headings become UPPERCASE with underlines, bold text is wrapped in *asterisks*, lists become indented with bullets (- ) or numbers (1.), horizontal rules become dashed lines - Convert HTML tables to aligned plain text tables with column padding and separator rows - Convert hyperlinks to "text [URL]" format, or optionally strip URLs via --no-urls flag - Preserve paragraph spacing: consecutive block elements get blank line separators - Handle HTML entities: decode & < > — etc. to their text equivalents - Support extracting text from only specific HTML elements via --selector flag (CSS selector syntax, e.g., --selector "article" or --selector ".content") - Support extracting and listing all URLs found in the document via --extract-urls flag - Set maximum line width via --width flag (default: 80 characters) with word wrapping - Support batch conversion of multiple HTML files via --batch flag - Print the plain text output to console by default - Save to a file via --output flag (default: same base name with .txt extension) - If no input is given, generate a sample HTML page with headings, paragraphs, links, tables, lists, images, inline styles, scripts, and HTML entities, then convert it and display both the original HTML and the extracted text - Handle errors: malformed HTML (parse gracefully), encoding detection, and binary file detection Create a complete C++ project for a clean Ubuntu 22.04 machine with only G++ 12+ and CMake 3.22+ installed. Include: - Source code - CMakeLists.txt with all dependencies (direct and transitive) pinned to exact versions - README.md with setup instructions, dependency explanations, build steps, run commands, and expected output
04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
3 files, exactly as written, before any repair.
CMakeLists.txt
cmake_minimum_required(VERSION 3.16) project(html_to_text_extractor LANGUAGES CXX) set(CMAKE_CXX_STANDARD 17) set(CMAKE_CXX_STANDARD_REQUIRED ON) add_executable(html_to_text_extractor src/main.cpp)
README.md
# HTML to Plain Text Extractor (C++) ## Requirements - Ubuntu 22.04 - CMake 3.16+ - C++17 compiler (g++/clang++) ## Build ```bash cmake -S . -B build cmake --build build ``` ## Run ```bash ./build/html_to_text_extractor input.html ./build/html_to_text_extractor --no-urls --extract-urls input.html ./build/html_to_text_extractor --selector ".content" --batch a.html b.html --output out_dir ``` If no input is passed, the program creates a sample HTML file and demonstrates conversion. ## Output - Extracted text written to `.txt` - Console report and optional URL list - JSON summary file: `html_extract_report.json`
src/main.cpp
#include <algorithm>
#include <chrono>
#include <filesystem>
#include <fstream>
#include <iomanip>
#include <iostream>
#include <regex>
#include <set>
#include <sstream>
#include <stdexcept>
#include <string>
#include <vector>
namespace fs = std::filesystem;
struct Config {
bool noUrls = false;
bool extractUrls = false;
bool batch = false;
std::string selector;
int width = 80;
std::string output;
std::vector<std::string> inputs;
};
struct ConvertResult {
std::string text;
std::vector<std::string> urls;
};
static std::string trim(const std::string& s) {
size_t b = s.find_first_not_of(" \t\r\n");
if (b == std::string::npos) return "";
size_t e = s.find_last_not_of(" \t\r\n");
return s.substr(b, e - b + 1);
}
static std::string readFile(const fs::path& p) {
std::ifstream in(p, std::ios::binary);
if (!in) throw std::runtime_error("Cannot open file: " + p.string());
std::ostringstream ss;
ss << in.rdbuf();
return ss.str();
}
static void writeFile(const fs::path& p, const std::string& content) {
std::ofstream out(p, std::ios::binary);
if (!out) throw std::runtime_error("Cannot write file: " + p.string());
out << content;
}
static bool isBinary(const fs::path& p) {
std::ifstream in(p, std::ios::binary);
char c;
for (int i = 0; i < 1024 && in.get(c); ++i) {
if (c == '\0') return true;
}
return false;
}
static std::string decodeEntities(std::string s) {
const std::vector<std::pair<std::string, std::string>> entities = {
{" ", " "}, {"&", "&"}, {"<", "<"}, {">", ">"},
{""", "\""}, {"'", "'"}, {"—", "-"}, {"–", "-"},
{" ", " "}, {"&", "&"}, {"<", "<"}, {">", ">"}
};
for (const auto& kv : entities) {
size_t pos = 0;
while ((pos = s.find(kv.first, pos)) != std::string::npos) {
s.replace(pos, kv.first.size(), kv.second);
pos += kv.second.size();
}
}
return s;
}
static std::string replaceBlocks(const std::string& input, const std::regex& rx,
const std::function<std::string(const std::smatch&)>& fn) {
std::string out;
size_t pos = 0;
for (std::sregex_iterator it(input.begin(), input.end(), rx), end; it != end; ++it) {
const std::smatch& m = *it;
out.append(input.substr(pos, static_cast<size_t>(m.position()) - pos));
out.append(fn(m));
pos = static_cast<size_t>(m.position() + m.length());
}
out.append(input.substr(pos));
return out;
}
static std::string stripNoise(std::string html) {
html = std::regex_replace(html, std::regex("<!--[\\s\\S]*?-->", std::regex::icase), "");
html = std::regex_replace(html, std::regex("<script\\b[^>]*>[\\s\\S]*?</script>", std::regex::icase), "");
html = std::regex_replace(html, std::regex("<style\\b[^>]*>[\\s\\S]*?</style>", std::regex::icase), "");
return html;
}
static std::vector<std::string> extractUrls(const std::string& html) {
std::set<std::string> urls;
std::regex rx("\\b(?:href|src)\\s*=\\s*['\"]([^'\"]+)['\"]", std::regex::icase);
for (std::sregex_iterator it(html.begin(), html.end(), rx), end; it != end; ++it) {
urls.insert((*it)[1].str());
}
return std::vector<std::string>(urls.begin(), urls.end());
}
static std::string applySelector(const std::string& html, const std::string& selector) {
if (selector.empty()) return html;
std::vector<std::string> parts;
std::stringstream ss(selector);
std::string part;
while (std::getline(ss, part, ',')) {
part = trim(part);
if (!part.empty()) parts.push_back(part);
}
std::vector<std::string> chunks;
for (const std::string& p : parts) {
std::regex rx("", std::regex::icase);
if (!p.empty() && p[0] == '.') {
std::string cls = p.substr(1);
rx = std::regex("<([a-z0-9]+)([^>]*class=['\"][^'\"]*\\b" + cls + "\\b[^'\"]*['\"][^>]*)>[\\s\\S]*?</\\1>", std::regex::icase);
} else if (!p.empty() && p[0] == '#') {
std::string id = p.substr(1);
rx = std::regex("<([a-z0-9]+)([^>]*id=['\"]" + id + "['\"][^>]*)>[\\s\\S]*?</\\1>", std::regex::icase);
} else {
rx = std::regex("<" + p + "\\b[^>]*>[\\s\\S]*?</" + p + ">", std::regex::icase);
}
for (std::sregex_iterator it(html.begin(), html.end(), rx), end; it != end; ++it) {
chunks.push_back((*it).str());
}
}
std::ostringstream out;
for (size_t i = 0; i < chunks.size(); ++i) {
if (i) out << "\n";
out << chunks[i];
}
return out.str();
}
static std::string renderList(const std::string& block, bool ordered) {
std::regex liRx("<li\\b[^>]*>([\\s\\S]*?)</li>", std::regex::icase);
std::ostringstream out;
out << "\n";
int idx = 1;
for (std::sregex_iterator it(block.begin(), block.end(), liRx), end; it != end; ++it) {
std::string t = std::regex_replace((*it)[1].str(), std::regex("<[^>]+>"), " ");
t = decodeEntities(trim(std::regex_replace(t, std::regex("\\s+"), " ")));
out << (ordered ? std::to_string(idx++) + ". " : "- ") << t << "\n";
}
out << "\n";
return out.str();
}
static std::string renderTable(const std::string& block) {
std::regex rowRx("<tr\\b[^>]*>([\\s\\S]*?)</tr>", std::regex::icase);
std::regex cellRx("<(?:th|td)\\b[^>]*>([\\s\\S]*?)</(?:th|td)>", std::regex::icase);
std::vector<std::vector<std::string>> rows;
for (std::sregex_iterator it(block.begin(), block.end(), rowRx), end; it != end; ++it) {
std::vector<std::string> row;
std::string rowHtml = (*it)[1].str();
for (std::sregex_iterator ci(rowHtml.begin(), rowHtml.end(), cellRx), cend; ci != cend; ++ci) {
std::string t = std::regex_replace((*ci)[1].str(), std::regex("<[^>]+>"), " ");
t = decodeEntities(trim(std::regex_replace(t, std::regex("\\s+"), " ")));
row.push_back(t);
}
if (!row.empty()) rows.push_back(row);
}
if (rows.empty()) return "";
size_t cols = 0;
for (const auto& r : rows) cols = std::max(cols, r.size());
std::vector<size_t> widths(cols, 0);
for (const auto& r : rows) {
for (size_t c = 0; c < cols; ++c) {
std::string v = c < r.size() ? r[c] : "";
widths[c] = std::max(widths[c], v.size());
}
}
std::ostringstream out;
out << "\n";
for (size_t ri = 0; ri < rows.size(); ++ri) {
out << "| ";
for (size_t c = 0; c < cols; ++c) {
std::string v = c < rows[ri].size() ? rows[ri][c] : "";
out << std::left << std::setw(static_cast<int>(widths[c])) << v;
out << (c + 1 == cols ? " |\n" : " | ");
}
if (ri == 0) {
out << "| ";
for (size_t c = 0; c < cols; ++c) {
out << std::string(std::max<size_t>(3, widths[c]), '-');
out << (c + 1 == cols ? " |\n" : " | ");
}
}
}
out << "\n";
return out.str();
}
static std::string wrapText(const std::string& text, int width) {
std::stringstream in(text);
std::string line;
std::ostringstream out;
bool firstLine = true;
while (std::getline(in, line)) {
if (!firstLine) out << "\n";
firstLine = false;
line = trim(line);
if (line.empty()) continue;
while (static_cast<int>(line.size()) > width) {
int cut = width;
for (int i = width; i >= 0; --i) {
if (line[static_cast<size_t>(i)] == ' ') { cut = i; break; }
}
out << line.substr(0, static_cast<size_t>(cut)) << "\n";
line = trim(line.substr(static_cast<size_t>(cut)));
}
out << line;
}
return out.str();
}
static ConvertResult convert(std::string html, const Config& cfg) {
html = stripNoise(html);
std::vector<std::string> urls = extractUrls(html);
html = applySelector(html, cfg.selector);
html = replaceBlocks(html, std::regex("<table\\b[^>]*>[\\s\\S]*?</table>", std::regex::icase), [](const std::smatch& m) {
return renderTable(m.str());
});
html = replaceBlocks(html, std::regex("<ol\\b[^>]*>[\\s\\S]*?</ol>", std::regex::icase), [](const std::smatch& m) {
return renderList(m.str(), true);
});
html = replaceBlocks(html, std::regex("<ul\\b[^>]*>[\\s\\S]*?</ul>", std::regex::icase), [](const std::smatch& m) {
return renderList(m.str(), false);
});
html = replaceBlocks(html, std::regex("<h([1-6])\\b[^>]*>([\\s\\S]*?)</h\\1>", std::regex::icase), [](const std::smatch& m) {
std::string t = std::regex_replace(m[2].str(), std::regex("<[^>]+>"), " ");
t = decodeEntities(trim(std::regex_replace(t, std::regex("\\s+"), " ")));
std::transform(t.begin(), t.end(), t.begin(), [](unsigned char c) { return static_cast<char>(std::toupper(c)); });
char fill = (m[1].str() == "1" || m[1].str() == "2") ? '=' : '-';
return "\n" + t + "\n" + std::string(std::max<size_t>(3, t.size()), fill) + "\n";
});
html = replaceBlocks(html, std::regex("<(?:strong|b)\\b[^>]*>([\\s\\S]*?)</(?:strong|b)>", std::regex::icase), [](const std::smatch& m) {
std::string t = std::regex_replace(m[1].str(), std::regex("<[^>]+>"), " ");
t = decodeEntities(trim(std::regex_replace(t, std::regex("\\s+"), " ")));
return "*" + t + "*";
});
html = std::regex_replace(html, std::regex("<hr\\b[^>]*?/?>", std::regex::icase), "\n----------------------------------------\n");
html = replaceBlocks(html, std::regex("<a\\b[^>]*href=['\"]([^'\"]+)['\"][^>]*>([\\s\\S]*?)</a>", std::regex::icase), [&cfg](const std::smatch& m) {
std::string href = m[1].str();
std::string t = std::regex_replace(m[2].str(), std::regex("<[^>]+>"), " ");
t = decodeEntities(trim(std::regex_replace(t, std::regex("\\s+"), " ")));
return cfg.noUrls ? t : t + " [" + href + "]";
});
html = std::regex_replace(html, std::regex("<br\\b[^>]*?/?>", std::regex::icase), "\n");
html = std::regex_replace(html, std::regex("</?(?:p|div|section|article|header|footer|main|aside|nav|figure|figcaption|blockquote)\\b[^>]*>", std::regex::icase), "\n\n");
html = std::regex_replace(html, std::regex("<[^>]+>"), " ");
html = decodeEntities(html);
html = std::regex_replace(html, std::regex("[ \t]+\\n"), "\n");
html = std::regex_replace(html, std::regex("\\n{3,}"), "\n\n");
html = std::regex_replace(html, std::regex("[ \t]{2,}"), " ");
html = trim(html);
html = wrapText(html, cfg.width);
ConvertResult result{html, urls};
return result;
}
static Config parseArgs(int argc, char** argv) {
Config cfg;
for (int i = 1; i < argc; ++i) {
std::string a = argv[i];
if (a == "--no-urls") cfg.noUrls = true;
else if (a == "--extract-urls") cfg.extractUrls = true;
else if (a == "--batch") cfg.batch = true;
else if (a == "--selector" && i + 1 < argc) cfg.selector = argv[++i];
else if (a == "--width" && i + 1 < argc) cfg.width = std::stoi(argv[++i]);
else if (a == "--output" && i + 1 < argc) cfg.output = argv[++i];
else if (a.rfind("--", 0) == 0) throw std::runtime_error("Unknown option: " + a);
else cfg.inputs.push_back(a);
}
if (cfg.width < 20) throw std::runtime_error("--width must be >= 20");
return cfg;
}
static std::string sampleHtml() {
return R"(<!doctype html>
<html><head><title>Sample</title><style>.x{color:red}</style><script>console.log('x')</script></head>
<body>
<article class='content' id='main'>
<h1>Demo Article</h1>
<p>This is a <strong>sample</strong> paragraph with <a href='https://example.com'>link</a> & entities — and spaces here.</p>
<h2>List</h2>
<ul><li>First</li><li>Second</li></ul>
<ol><li>One</li><li>Two</li></ol>
<hr>
<h2>Table</h2>
<table><tr><th>Name</th><th>Score</th></tr><tr><td>Ada</td><td>98</td></tr><tr><td>Linus</td><td>95</td></tr></table>
<img src='https://cdn.example.com/image.png'>
</article>
</body></html>)";
}
int main(int argc, char** argv) {
try {
Config cfg = parseArgs(argc, argv);
if (cfg.inputs.empty()) {
writeFile("sample_page.html", sampleHtml());
ConvertResult r = convert(sampleHtml(), cfg);
writeFile("sample_page.txt", r.text + "\n");
std::cout << "===== ORIGINAL HTML =====\n" << sampleHtml() << "\n\n===== EXTRACTED TEXT =====\n" << r.text << "\n";
if (cfg.extractUrls) {
std::cout << "\nURLs:\n";
for (const auto& u : r.urls) std::cout << "- " << u << "\n";
}
return 0;
}
std::vector<std::string> targets = cfg.batch ? cfg.inputs : std::vector<std::string>{cfg.inputs.front()};
if (cfg.batch && !cfg.output.empty()) fs::create_directories(cfg.output);
std::ostringstream report;
report << "{\n \"files\": [\n";
bool first = true;
for (const auto& input : targets) {
fs::path p = input;
if (!fs::exists(p)) {
std::cerr << "Warning: missing file " << input << "\n";
continue;
}
try {
if (isBinary(p)) throw std::runtime_error("Binary file detected: " + input);
ConvertResult r = convert(readFile(p), cfg);
fs::path out;
if (cfg.batch) {
fs::path dir = cfg.output.empty() ? p.parent_path() : fs::path(cfg.output);
out = dir / (p.stem().string() + ".txt");
} else {
out = cfg.output.empty() ? (p.parent_path() / (p.stem().string() + ".txt")) : fs::path(cfg.output);
}
writeFile(out, r.text + "\n");
std::cout << r.text << "\n";
if (cfg.extractUrls) {
std::cout << "\nURLs for " << input << ":\n";
for (const auto& u : r.urls) std::cout << "- " << u << "\n";
}
if (!first) report << ",\n";
first = false;
report << " {\"input\":\"" << input << "\",\"output\":\"" << out.string() << "\",\"urlsCount\":" << r.urls.size() << "}";
} catch (const std::exception& ex) {
std::cerr << "Warning: " << ex.what() << "\n";
}
}
report << "\n ]\n}\n";
writeFile("html_extract_report.json", report.str());
} catch (const std::exception& ex) {
std::cerr << "Error: " << ex.what() << "\n";
return 1;
}
return 0;
}