Multi-Format Data Converter (javascript, written by Codex)
envgap__codex__javascript-t1-8
Written by a coding agent; not on GitHubWritten 2026-03-02
01 / FAILURE SIGNATURE
As the study recorded it
None
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
package.json- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
codex/javascript-t1 #8 · read the task the agent was given
Codex wrote this javascript project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written.
Task given to the agent:
TASK: Multi-Format Data Converter
Write a program that converts data files between CSV, JSON, XML, and YAML formats while preserving data types, nested structures, and handling large files efficiently.
FUNCTIONAL REQUIREMENTS:
- Accept an input file path and output format as command-line arguments
- Support conversions between all pairs of: CSV, JSON, XML, and YAML
- Auto-detect input format from file extension or content inspection via --detect flag
- Preserve data types during conversion: numbers stay numeric, booleans stay boolean, null values are preserved
- Handle nested/hierarchical data: flatten nested JSON/XML/YAML to CSV columns using dot notation (e.g., address.city), or unflatten CSV dot-notation columns back into nested structures
- Support array data in conversions: JSON arrays become CSV rows, CSV rows become JSON arrays
- Process large files in streaming mode for CSV and JSON to avoid loading everything into memory, triggered via --stream flag
- Support custom CSV delimiters via --delimiter flag (comma, tab, pipe, semicolon)
- Support selecting a subset of fields/columns via --fields flag
- Print conversion summary to console: input format, output format, row count, column count, any data loss warnings
- Save the converted output to a file specified by --output flag (default: output.{format})
- If no input file is given, generate a sample dataset with nested objects, arrays, mixed types, and null values in JSON format, then convert it to all other formats
- Handle encoding differences (UTF-8, Latin-1) and BOM markers gracefully
Create a complete JavaScript project for a clean Ubuntu 22.04 machine with only Node.js 20+ (LTS) installed. Include:
- Source code
- package.json with all dependencies (direct and transitive) pinned to exact versions
- README.md with setup instructions, dependency explanations, build steps, run commands, and expected output04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
4 files, exactly as written, before any repair.
package-lock.json
{
"name": "multi-format-data-converter",
"version": "1.0.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "multi-format-data-converter",
"version": "1.0.0",
"dependencies": {
"fast-xml-parser": "4.5.1",
"js-yaml": "4.1.0"
}
},
"node_modules/argparse": {
"version": "2.0.1"
},
"node_modules/fast-xml-parser": {
"version": "4.5.1",
"dependencies": {
"strnum": "1.0.5"
}
},
"node_modules/js-yaml": {
"version": "4.1.0",
"dependencies": {
"argparse": "2.0.1"
}
},
"node_modules/strnum": {
"version": "1.0.5"
}
}
}
package.json
{
"name": "multi-format-data-converter",
"version": "1.0.0",
"description": "Convert CSV, JSON, XML, and YAML files with nested structure handling and type preservation.",
"type": "module",
"main": "src/index.js",
"scripts": {
"start": "node src/index.js"
},
"engines": {
"node": ">=20.0.0"
},
"dependencies": {
"fast-xml-parser": "4.5.1",
"js-yaml": "4.1.0"
}
}
README.md
# Multi-Format Data Converter (JavaScript) Converts data between `CSV`, `JSON`, `XML`, and `YAML` with type preservation, nested flatten/unflatten logic, and optional stream mode. ## Requirements - Ubuntu 22.04 - Node.js 20+ (LTS) ## Dependencies - Direct: - `fast-xml-parser@4.5.1` - `js-yaml@4.1.0` - Transitive: - `argparse@2.0.1` - `strnum@1.0.5` All versions are pinned in `package.json` and `package-lock.json`. ## Setup ```bash npm install ``` ## Run General form: ```bash node src/index.js <input-file> <output-format> [--output file] [--detect] [--stream] [--delimiter comma|tab|pipe|semicolon] [--fields a,b,c] ``` Examples: ```bash node src/index.js ./data.json csv --output out.csv --fields id,name,address.city node src/index.js ./data.csv json --delimiter semicolon --detect node src/index.js ./data.ndjson yaml --stream ``` No input (generates sample JSON and converts to all other formats): ```bash node src/index.js ``` ## Notes - Dot notation is used when flattening nested data to CSV columns. - CSV dot-notation columns are unflattened back to nested objects. - Stream mode supports CSV and NDJSON-style JSON input. - Handles UTF-8 BOM and Latin-1 decoding fallback.
src/index.js
import fs from "fs";
import path from "path";
import readline from "readline";
import yaml from "js-yaml";
import { XMLBuilder, XMLParser } from "fast-xml-parser";
const FORMAT_SET = new Set(["csv", "json", "xml", "yaml"]);
function parseArgs(argv) {
const options = {
delimiter: ",",
};
const positional = [];
for (let i = 0; i < argv.length; i += 1) {
const token = argv[i];
if (token.startsWith("--")) {
const key = token.slice(2);
const next = argv[i + 1];
if (next && !next.startsWith("--")) {
options[key] = next;
i += 1;
} else {
options[key] = true;
}
} else {
positional.push(token);
}
}
return { options, positional };
}
function normalizeDelimiter(value) {
if (!value) return ",";
if (value === "," || value === "comma") return ",";
if (value === "\\t" || value === "tab") return "\t";
if (value === "|" || value === "pipe") return "|";
if (value === ";" || value === "semicolon") return ";";
return value[0];
}
function stripBom(text) {
return text.replace(/^\uFEFF/, "");
}
function readTextWithEncoding(filePath) {
const raw = fs.readFileSync(filePath);
const utf8 = raw.toString("utf8");
const bad = (utf8.match(/\uFFFD/g) || []).length;
if (bad > 0) {
return { text: stripBom(raw.toString("latin1")), encoding: "latin1" };
}
return { text: stripBom(utf8), encoding: "utf8" };
}
function detectFormat(inputPath, text, detectFlag) {
if (!detectFlag && inputPath) {
const ext = path.extname(inputPath).toLowerCase();
if (ext === ".csv") return "csv";
if (ext === ".json" || ext === ".jsonl" || ext === ".ndjson") return "json";
if (ext === ".xml") return "xml";
if (ext === ".yaml" || ext === ".yml") return "yaml";
}
const trimmed = text.trim();
if (trimmed.startsWith("{") || trimmed.startsWith("[") || trimmed.split("\n").every((l) => l.trim() === "" || l.trim().startsWith("{"))) return "json";
if (trimmed.startsWith("<")) return "xml";
if (/^\s*[\w.-]+\s*:/.test(trimmed)) return "yaml";
return "csv";
}
function parsePrimitive(value) {
if (value == null) return null;
const s = String(value).trim();
if (s === "") return null;
if (/^null$/i.test(s)) return null;
if (/^(true|false)$/i.test(s)) return /^true$/i.test(s);
if (/^[-+]?\d+$/.test(s)) return Number.parseInt(s, 10);
if (/^[-+]?\d+\.\d+$/.test(s)) return Number.parseFloat(s);
return value;
}
function parseCsvLine(line, delimiter) {
const out = [];
let current = "";
let inQuotes = false;
for (let i = 0; i < line.length; i += 1) {
const ch = line[i];
if (ch === '"') {
if (inQuotes && line[i + 1] === '"') {
current += '"';
i += 1;
} else {
inQuotes = !inQuotes;
}
} else if (ch === delimiter && !inQuotes) {
out.push(current);
current = "";
} else {
current += ch;
}
}
out.push(current);
return out;
}
function csvEscape(value, delimiter) {
const text = value == null ? "" : String(value);
if (text.includes(delimiter) || text.includes('"') || text.includes("\n")) {
return `"${text.replace(/"/g, "\"\"")}"`;
}
return text;
}
function flattenObject(value, prefix = "", out = {}) {
if (value == null) {
if (prefix) out[prefix] = null;
return out;
}
if (Array.isArray(value)) {
if (value.length === 0) {
if (prefix) out[prefix] = [];
return out;
}
for (let i = 0; i < value.length; i += 1) {
const key = prefix ? `${prefix}.${i}` : String(i);
flattenObject(value[i], key, out);
}
return out;
}
if (typeof value === "object") {
const keys = Object.keys(value);
if (keys.length === 0 && prefix) {
out[prefix] = {};
return out;
}
for (const key of keys) {
const next = prefix ? `${prefix}.${key}` : key;
flattenObject(value[key], next, out);
}
return out;
}
if (prefix) out[prefix] = value;
return out;
}
function assignPath(obj, parts, value) {
let current = obj;
for (let i = 0; i < parts.length; i += 1) {
const part = parts[i];
const isLast = i === parts.length - 1;
const index = /^[0-9]+$/.test(part) ? Number.parseInt(part, 10) : null;
if (isLast) {
if (index != null) {
if (!Array.isArray(current)) {
// Convert object placeholder to array-like container.
// This path is only hit when malformed mixed schema appears.
current[part] = value;
} else {
current[index] = value;
}
} else {
current[part] = value;
}
return;
}
const nextPart = parts[i + 1];
const nextIsIndex = /^[0-9]+$/.test(nextPart);
if (index != null) {
if (!Array.isArray(current)) {
current[part] = current[part] ?? (nextIsIndex ? [] : {});
current = current[part];
} else {
if (current[index] == null) current[index] = nextIsIndex ? [] : {};
current = current[index];
}
} else {
if (current[part] == null) current[part] = nextIsIndex ? [] : {};
current = current[part];
}
}
}
function unflattenRow(flatRow) {
const out = {};
for (const [key, value] of Object.entries(flatRow)) {
if (!key.includes(".")) {
out[key] = value;
} else {
assignPath(out, key.split("."), value);
}
}
return out;
}
function normalizeRows(data) {
if (Array.isArray(data)) return data.map((v) => (v && typeof v === "object" ? v : { value: v }));
if (data && typeof data === "object") return [data];
return [{ value: data }];
}
function applyFields(rows, fields) {
if (!fields || fields.length === 0) return rows;
return rows.map((row) => {
const flat = flattenObject(row);
const selected = {};
for (const f of fields) {
if (Object.prototype.hasOwnProperty.call(flat, f)) selected[f] = flat[f];
}
return unflattenRow(selected);
});
}
function countColumns(rows) {
const columns = new Set();
for (const row of rows) {
for (const key of Object.keys(flattenObject(row))) columns.add(key);
}
return columns.size;
}
async function readCsvStream(filePath, delimiter) {
const stream = fs.createReadStream(filePath, { encoding: "utf8" });
const rl = readline.createInterface({ input: stream, crlfDelay: Infinity });
let headers = null;
const rows = [];
for await (const rawLine of rl) {
const line = stripBom(rawLine);
if (!line.trim()) continue;
if (!headers) {
headers = parseCsvLine(line, delimiter).map((h) => h.trim());
continue;
}
const parts = parseCsvLine(line, delimiter);
const flat = {};
for (let i = 0; i < headers.length; i += 1) flat[headers[i]] = parsePrimitive(parts[i] ?? "");
rows.push(unflattenRow(flat));
}
return rows;
}
async function readJsonStream(filePath) {
const stream = fs.createReadStream(filePath, { encoding: "utf8" });
const rl = readline.createInterface({ input: stream, crlfDelay: Infinity });
const rows = [];
for await (const line of rl) {
const trimmed = line.trim();
if (!trimmed) continue;
if (!(trimmed.startsWith("{") && trimmed.endsWith("}"))) {
throw new Error("Stream mode for JSON expects NDJSON (one JSON object per line).");
}
rows.push(JSON.parse(trimmed));
}
return rows;
}
async function parseInput(filePath, format, delimiter, streamMode, warnings) {
if (streamMode && format === "csv") {
return readCsvStream(filePath, delimiter);
}
if (streamMode && format === "json") {
try {
return await readJsonStream(filePath);
} catch (error) {
warnings.push("JSON stream mode works best with NDJSON. Falling back to full parse.");
}
}
const { text } = readTextWithEncoding(filePath);
if (format === "csv") {
const lines = text.replace(/\r\n/g, "\n").replace(/\r/g, "\n").split("\n").filter((l) => l.trim());
if (lines.length === 0) return [];
const headers = parseCsvLine(lines[0], delimiter).map((h) => h.trim());
return lines.slice(1).map((line) => {
const parts = parseCsvLine(line, delimiter);
const flat = {};
for (let i = 0; i < headers.length; i += 1) flat[headers[i]] = parsePrimitive(parts[i] ?? "");
return unflattenRow(flat);
});
}
if (format === "json") {
const parsed = JSON.parse(text);
return normalizeRows(parsed);
}
if (format === "yaml") {
const parsed = yaml.load(text);
return normalizeRows(parsed);
}
if (format === "xml") {
const parser = new XMLParser({
ignoreAttributes: false,
parseTagValue: true,
trimValues: true,
});
const parsed = parser.parse(text);
if (parsed.root && Array.isArray(parsed.root.item)) return normalizeRows(parsed.root.item);
if (parsed.root && parsed.root.item) return normalizeRows(parsed.root.item);
return normalizeRows(parsed);
}
throw new Error(`Unsupported input format: ${format}`);
}
function serializeCsv(rows, delimiter) {
const flatRows = rows.map((r) => flattenObject(r));
const headers = [...new Set(flatRows.flatMap((r) => Object.keys(r)))];
const lines = [headers.map((h) => csvEscape(h, delimiter)).join(delimiter)];
for (const row of flatRows) {
lines.push(headers.map((h) => csvEscape(row[h], delimiter)).join(delimiter));
}
return `${lines.join("\n")}\n`;
}
function serializeJson(rows) {
return `${JSON.stringify(rows, null, 2)}\n`;
}
function serializeYaml(rows) {
return `${yaml.dump(rows, { noRefs: true, lineWidth: 120 })}`;
}
function serializeXml(rows) {
const builder = new XMLBuilder({
ignoreAttributes: false,
suppressEmptyNode: false,
format: true,
});
return builder.build({ root: { item: rows } });
}
function writeOutput(outputPath, format, rows, delimiter) {
let content = "";
if (format === "csv") content = serializeCsv(rows, delimiter);
else if (format === "json") content = serializeJson(rows);
else if (format === "yaml") content = serializeYaml(rows);
else if (format === "xml") content = serializeXml(rows);
else throw new Error(`Unsupported output format: ${format}`);
fs.writeFileSync(outputPath, content, "utf8");
}
function printSummary(inputFormat, outputFormat, rows, warnings) {
console.log("Conversion Summary");
console.log("==================");
console.log(`Input format : ${inputFormat}`);
console.log(`Output format: ${outputFormat}`);
console.log(`Row count : ${rows.length}`);
console.log(`Column count : ${countColumns(rows)}`);
if (warnings.length === 0) {
console.log("Warnings : none");
} else {
console.log(`Warnings : ${warnings.join(" | ")}`);
}
}
function buildSampleData() {
return [
{
id: 1,
name: "Alice",
active: true,
score: 97.5,
address: { city: "Austin", zip: "73301" },
tags: ["premium", "beta"],
orders: [
{ id: "o1", amount: 39.95 },
{ id: "o2", amount: 12.0 },
],
last_login: null,
},
{
id: 2,
name: "Bob",
active: false,
score: 88,
address: { city: "Berlin", zip: "10115" },
tags: ["standard"],
orders: [{ id: "o3", amount: 120.1 }],
last_login: "2026-01-01T10:00:00Z",
},
];
}
async function convertSingle(inputPath, outputFormat, options) {
const delimiter = normalizeDelimiter(options.delimiter);
const warnings = [];
const detectFlag = Boolean(options.detect);
const streamMode = Boolean(options.stream);
const fields = options.fields ? options.fields.split(",").map((x) => x.trim()).filter(Boolean) : [];
const { text, encoding } = readTextWithEncoding(inputPath);
let inputFormat = detectFormat(inputPath, text, detectFlag);
if (!FORMAT_SET.has(inputFormat)) throw new Error(`Could not detect supported input format for ${inputPath}`);
if (!FORMAT_SET.has(outputFormat)) throw new Error(`Unsupported output format: ${outputFormat}`);
if (encoding !== "utf8") warnings.push(`Input decoded as ${encoding}`);
let rows = await parseInput(inputPath, inputFormat, delimiter, streamMode, warnings);
rows = applyFields(rows, fields);
if (outputFormat === "csv") {
warnings.push("Nested objects/arrays are flattened to dot-notation columns.");
}
const outputPath = path.resolve(options.output || `output.${outputFormat}`);
writeOutput(outputPath, outputFormat, rows, delimiter);
printSummary(inputFormat, outputFormat, rows, warnings);
console.log(`Output file : ${outputPath}`);
}
function writeSampleJson(filePath, data) {
fs.writeFileSync(filePath, `${JSON.stringify(data, null, 2)}\n`, "utf8");
}
async function main() {
const { options, positional } = parseArgs(process.argv.slice(2));
if (positional.length === 0) {
const sample = buildSampleData();
const samplePath = path.resolve("sample_data.json");
writeSampleJson(samplePath, sample);
console.log(`No input provided. Generated sample dataset: ${samplePath}`);
for (const fmt of ["csv", "xml", "yaml"]) {
const outputPath = path.resolve(`output.${fmt}`);
const warnings = [];
writeOutput(outputPath, fmt, sample, normalizeDelimiter(options.delimiter));
printSummary("json", fmt, sample, warnings);
console.log(`Output file : ${outputPath}`);
console.log("");
}
return;
}
const inputPath = path.resolve(positional[0]);
if (!fs.existsSync(inputPath)) {
console.error(`Input file not found: ${inputPath}`);
process.exit(1);
}
const outputFormat = (positional[1] || options.format || "").toLowerCase();
if (!FORMAT_SET.has(outputFormat)) {
console.error("Provide output format as second argument or --format (csv|json|xml|yaml).");
process.exit(1);
}
await convertSingle(inputPath, outputFormat, options);
}
main().catch((error) => {
console.error(`Conversion failed: ${error.message}`);
process.exit(1);
});