Data Compression Benchmark (javascript, written by Codex)
envgap__codex__javascript-t1-40
Written by a coding agent; not on GitHubWritten 2026-03-03
01 / FAILURE SIGNATURE
As the study recorded it
None
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
package.json- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
codex/javascript-t1 #40 · read the task the agent was given
Codex wrote this javascript project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written. Task given to the agent: TASK: Data Compression Benchmark Write a program that benchmarks multiple compression algorithms on given data files, comparing compression ratio, speed, memory usage, and decompression speed across algorithms and compression levels. FUNCTIONAL REQUIREMENTS: - Accept one or more file paths as command-line arguments to use as benchmark data - Support benchmarking multiple compression algorithms: DEFLATE (gzip), bzip2, LZMA (xz), LZ4 (if available), and zlib at various compression levels - For each algorithm, test at multiple compression levels (e.g., levels 1, 5, 9 for gzip) - Measure and report for each combination: compression ratio (compressed/original), compression speed (MB/s), decompression speed (MB/s), peak memory usage, and wall-clock time - Run each benchmark multiple times (configurable via --iterations flag, default 3) and report min/mean/max for timing measurements - Support a --quick flag to test only the default compression level for each algorithm - Generate a summary comparison table sorted by a configurable metric via --sort flag (ratio, compress-speed, decompress-speed; default: ratio) - Verify data integrity: decompress each result and verify it matches the original via checksum comparison - Support benchmarking with different data types via --generate flag: text (English prose), csv (tabular data), json (structured data), binary (random bytes), and mixed - Print results as a formatted table to console - Save the full benchmark report as JSON with --output flag (default: compression_benchmark.json) - If no input files are given, generate sample data files of each type (1MB each), benchmark all algorithms on each, and display a comprehensive comparison matrix - Handle errors: unsupported algorithms on the platform, out-of-memory during compression, and algorithm-specific limitations Create a complete JavaScript project for a clean Ubuntu 22.04 machine with only Node.js 20+ (LTS) installed. Include: - Source code - package.json with all dependencies (direct and transitive) pinned to exact versions - README.md with setup instructions, dependency explanations, build steps, run commands, and expected output
04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
4 files, exactly as written, before any repair.
package-lock.json
{
"name": "data-compression-benchmark",
"version": "1.0.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "data-compression-benchmark",
"version": "1.0.0",
"license": "MIT"
}
}
}
package.json
{
"name": "data-compression-benchmark",
"version": "1.0.0",
"description": "Benchmark compression algorithms",
"type": "module",
"main": "src/index.js",
"scripts": { "start": "node src/index.js" },
"engines": { "node": ">=20.0.0" },
"license": "MIT"
}
README.md
# Data Compression Benchmark (JavaScript) ## Requirements - Ubuntu 22.04 - Node.js 20+ - System tools for optional algorithms: `bzip2`, `xz`, `lz4` ## Install ```bash npm install ``` ## Run ```bash node src/index.js data1.bin data2.txt node src/index.js data.bin --iterations 5 --sort compress-speed node src/index.js --quick --generate mixed --output benchmark.json ``` If no inputs are provided, 1MB sample files are generated and benchmarked.
src/index.js
import fs from "node:fs";
import path from "node:path";
import os from "node:os";
import crypto from "node:crypto";
import zlib from "node:zlib";
import { execSync } from "node:child_process";
import { performance } from "node:perf_hooks";
function parseArgs(argv) {
const cfg = {
files: [],
iterations: 3,
quick: false,
sort: "ratio",
generate: null,
output: "compression_benchmark.json"
};
for (let i = 0; i < argv.length; i += 1) {
const a = argv[i];
if (!a.startsWith("--")) cfg.files.push(a);
else if (a === "--iterations") cfg.iterations = Number.parseInt(argv[++i], 10);
else if (a === "--quick") cfg.quick = true;
else if (a === "--sort") cfg.sort = argv[++i];
else if (a === "--generate") cfg.generate = argv[++i];
else if (a === "--output") cfg.output = argv[++i];
else throw new Error(`Unknown option: ${a}`);
}
return cfg;
}
function cmdExists(cmd) {
try {
execSync(`command -v ${cmd}`, { stdio: "ignore" });
return true;
} catch {
return false;
}
}
function sha256Buffer(buf) {
return crypto.createHash("sha256").update(buf).digest("hex");
}
function sha256File(file) {
return sha256Buffer(fs.readFileSync(file));
}
function humanMBps(bytes, sec) {
return sec > 0 ? (bytes / (1024 * 1024)) / sec : 0;
}
function stats(arr) {
const sorted = [...arr].sort((a, b) => a - b);
const mean = arr.reduce((s, x) => s + x, 0) / (arr.length || 1);
return { min: sorted[0] ?? 0, mean, max: sorted[sorted.length - 1] ?? 0 };
}
function generateSampleData(type, outDir, bytes = 1024 * 1024) {
fs.mkdirSync(outDir, { recursive: true });
const files = [];
const write = (name, content) => {
const p = path.join(outDir, name);
fs.writeFileSync(p, content);
files.push(p);
};
if (!type || type === "text" || type === "mixed") {
const line = "The quick brown fox jumps over the lazy dog. ";
write("sample_text.txt", line.repeat(Math.ceil(bytes / line.length)).slice(0, bytes));
}
if (!type || type === "csv" || type === "mixed") {
let csv = "id,name,value\n";
for (let i = 0; csv.length < bytes; i += 1) csv += `${i},user${i},${Math.sin(i).toFixed(6)}\n`;
write("sample_csv.csv", csv.slice(0, bytes));
}
if (!type || type === "json" || type === "mixed") {
const rows = [];
for (let i = 0; i < 6000; i += 1) rows.push({ id: i, ok: i % 2 === 0, value: i * 1.25, tag: `t${i % 10}` });
const s = JSON.stringify({ rows });
write("sample_json.json", s.repeat(Math.ceil(bytes / s.length)).slice(0, bytes));
}
if (!type || type === "binary" || type === "mixed") {
write("sample_bin.bin", crypto.randomBytes(bytes));
}
return files;
}
function compressBuiltin(buf, algo, level) {
if (algo === "gzip") return zlib.gzipSync(buf, { level });
if (algo === "zlib") return zlib.deflateSync(buf, { level });
throw new Error(`Unsupported builtin algo ${algo}`);
}
function decompressBuiltin(buf, algo) {
if (algo === "gzip") return zlib.gunzipSync(buf);
if (algo === "zlib") return zlib.inflateSync(buf);
throw new Error(`Unsupported builtin algo ${algo}`);
}
function runShellCompression(inputPath, algo, level, tmpDir) {
const outPath = path.join(tmpDir, `${path.basename(inputPath)}.${algo}.cmp`);
if (algo === "bzip2") execSync(`bzip2 -${Math.max(1, Math.min(9, level))} -c "${inputPath}" > "${outPath}"`);
else if (algo === "xz") execSync(`xz -${Math.max(0, Math.min(9, level))} -c "${inputPath}" > "${outPath}"`);
else if (algo === "lz4") execSync(`lz4 -${Math.max(1, Math.min(12, level))} -c "${inputPath}" > "${outPath}"`);
else throw new Error(`Unknown shell algo ${algo}`);
return outPath;
}
function runShellDecompression(compPath, algo, tmpDir) {
const outPath = path.join(tmpDir, `${path.basename(compPath)}.dec`);
if (algo === "bzip2") execSync(`bzip2 -dc "${compPath}" > "${outPath}"`);
else if (algo === "xz") execSync(`xz -dc "${compPath}" > "${outPath}"`);
else if (algo === "lz4") execSync(`lz4 -dc "${compPath}" > "${outPath}"`);
else throw new Error(`Unknown shell algo ${algo}`);
return outPath;
}
function benchmarkOne(file, algo, level, iterations) {
const src = fs.readFileSync(file);
const srcHash = sha256Buffer(src);
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), "bench-"));
const compTimes = [];
const decompTimes = [];
const compSizes = [];
const memDeltas = [];
try {
for (let i = 0; i < iterations; i += 1) {
const mem0 = process.memoryUsage().rss;
let compressed;
let compPath;
const c0 = performance.now();
if (algo === "gzip" || algo === "zlib") {
compressed = compressBuiltin(src, algo, level);
} else {
compPath = runShellCompression(file, algo, level, tmpDir);
compressed = fs.readFileSync(compPath);
}
const cSec = (performance.now() - c0) / 1000;
compTimes.push(cSec);
compSizes.push(compressed.length);
const d0 = performance.now();
let decompressed;
if (algo === "gzip" || algo === "zlib") {
decompressed = decompressBuiltin(compressed, algo);
} else {
const decPath = runShellDecompression(compPath, algo, tmpDir);
decompressed = fs.readFileSync(decPath);
}
const dSec = (performance.now() - d0) / 1000;
decompTimes.push(dSec);
const mem1 = process.memoryUsage().rss;
memDeltas.push(Math.max(0, mem1 - mem0));
if (sha256Buffer(decompressed) !== srcHash) throw new Error(`Integrity check failed for ${algo} level ${level}`);
}
} finally {
fs.rmSync(tmpDir, { recursive: true, force: true });
}
const cStats = stats(compTimes);
const dStats = stats(decompTimes);
const avgComp = compSizes.reduce((s, x) => s + x, 0) / compSizes.length;
const ratio = avgComp / src.length;
return {
file,
algorithm: algo,
level,
compressionRatio: Number(ratio.toFixed(6)),
compressionSize: Math.round(avgComp),
compressionTime: { min: cStats.min, mean: cStats.mean, max: cStats.max },
decompressionTime: { min: dStats.min, mean: dStats.mean, max: dStats.max },
compressSpeedMBps: Number(humanMBps(src.length, cStats.mean).toFixed(3)),
decompressSpeedMBps: Number(humanMBps(src.length, dStats.mean).toFixed(3)),
peakMemoryBytes: Math.max(...memDeltas)
};
}
function algorithmPlans(quick) {
const plans = [];
const baseLevels = quick ? [6] : [1, 5, 9];
plans.push({ algo: "gzip", levels: baseLevels, available: true, reason: null });
plans.push({ algo: "zlib", levels: baseLevels, available: true, reason: null });
plans.push({ algo: "bzip2", levels: quick ? [6] : [1, 5, 9], available: cmdExists("bzip2"), reason: "missing bzip2" });
plans.push({ algo: "xz", levels: quick ? [6] : [0, 6, 9], available: cmdExists("xz"), reason: "missing xz" });
plans.push({ algo: "lz4", levels: [1], available: cmdExists("lz4"), reason: "missing lz4" });
return plans;
}
function printTable(rows, sortMetric) {
const metricKey = sortMetric === "compress-speed" ? "compressSpeedMBps" : sortMetric === "decompress-speed" ? "decompressSpeedMBps" : "compressionRatio";
rows.sort((a, b) => metricKey.includes("Ratio") ? a[metricKey] - b[metricKey] : b[metricKey] - a[metricKey]);
const headers = ["File", "Algo", "Level", "Ratio", "Comp MB/s", "Decomp MB/s", "PeakMem"];
const data = rows.map((r) => [path.basename(r.file), r.algorithm, String(r.level), r.compressionRatio.toFixed(4), r.compressSpeedMBps.toFixed(2), r.decompressSpeedMBps.toFixed(2), String(r.peakMemoryBytes)]);
const widths = headers.map((h, i) => Math.max(h.length, ...data.map((d) => d[i].length)));
const line = (arr) => arr.map((v, i) => v.padEnd(widths[i], " ")).join(" ");
console.log(line(headers));
console.log(line(widths.map((w) => "-".repeat(w))));
data.forEach((d) => console.log(line(d)));
}
function main() {
try {
const cfg = parseArgs(process.argv.slice(2));
let files = [...cfg.files];
if (!files.length) files = generateSampleData(null, path.resolve("sample_compression_data"));
if (cfg.generate) files = generateSampleData(cfg.generate, path.resolve("generated_compression_data"));
const plans = algorithmPlans(cfg.quick);
const skipped = [];
const results = [];
for (const file of files) {
for (const p of plans) {
if (!p.available) {
skipped.push({ file, algorithm: p.algo, reason: p.reason });
continue;
}
for (const level of p.levels) {
try {
results.push(benchmarkOne(file, p.algo, level, cfg.iterations));
console.log(`bench ok: ${path.basename(file)} ${p.algo}@${level}`);
} catch (err) {
skipped.push({ file, algorithm: p.algo, level, reason: err.message });
console.error(`bench fail: ${path.basename(file)} ${p.algo}@${level}: ${err.message}`);
}
}
}
}
printTable(results, cfg.sort);
const report = {
generatedAt: new Date().toISOString(),
config: cfg,
files,
results,
skipped
};
fs.writeFileSync(path.resolve(cfg.output), JSON.stringify(report, null, 2), "utf8");
} catch (err) {
console.error(`Error: ${err.message}`);
process.exit(1);
}
}
main();