Data Profiling Tool (javascript, written by Codex)
envgap__codex__javascript-t1-7
Written by a coding agent; not on GitHubWritten 2026-03-02
01 / FAILURE SIGNATURE
As the study recorded it
None
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
package.json- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
codex/javascript-t1 #7 · read the task the agent was given
Codex wrote this javascript project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written. Task given to the agent: TASK: Data Profiling Tool Write a program that profiles tabular datasets by analyzing column types, distributions, missing values, correlations, and data quality issues, producing a comprehensive data quality report. FUNCTIONAL REQUIREMENTS: - Accept a CSV or JSON data file path as a command-line argument - Auto-detect column data types: numeric (integer vs float), string, boolean, date/timestamp, and categorical (low cardinality strings) - For numeric columns: compute min, max, mean, median, standard deviation, skewness, and percentiles (25th, 50th, 75th, 95th, 99th) - For string columns: compute min/max/average length, most common values (top 10), and unique count - For all columns: count total values, missing/null values, missing percentage, and unique value count - Detect potential data quality issues: columns that are entirely null, columns with a single unique value, numeric columns with extreme outliers (beyond 4 standard deviations), and string columns that look like misclassified numbers or dates - Compute a pairwise correlation matrix for all numeric columns - Print a formatted summary report to console showing key statistics per column - Save the full profiling report as a JSON file with --output flag (default: data_profile.json) - If no input file is given, generate a sample dataset with 1000 rows across at least 8 columns of mixed types including intentional quality issues, then profile it - Handle files with inconsistent delimiters or encoding issues gracefully Create a complete JavaScript project for a clean Ubuntu 22.04 machine with only Node.js 20+ (LTS) installed. Include: - Source code - package.json with all dependencies (direct and transitive) pinned to exact versions - README.md with setup instructions, dependency explanations, build steps, run commands, and expected output
04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
4 files, exactly as written, before any repair.
package-lock.json
{
"name": "data-profiling-tool",
"version": "1.0.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "data-profiling-tool",
"version": "1.0.0",
"dependencies": {}
}
}
}
package.json
{
"name": "data-profiling-tool",
"version": "1.0.0",
"description": "Profiles CSV/JSON datasets and generates data quality reports.",
"type": "module",
"main": "src/index.js",
"scripts": {
"start": "node src/index.js"
},
"engines": {
"node": ">=20.0.0"
},
"dependencies": {}
}
README.md
# Data Profiling Tool (JavaScript) Profiles CSV/JSON tabular data for types, distributions, missingness, outliers, correlations, and quality issues. ## Requirements - Ubuntu 22.04 - Node.js 20+ (LTS) ## Dependencies - Direct: none - Transitive: none Pinned in `package.json` and `package-lock.json`. ## Run With input file: ```bash node src/index.js /path/to/data.csv --output data_profile.json ``` JSON input: ```bash node src/index.js /path/to/data.json --output data_profile.json ``` No input (generates sample dataset with intentional quality issues): ```bash node src/index.js ``` ## Output - Console summary table per column - Full JSON report (`data_profile.json` by default) ## Notes - Handles CSV delimiter detection across `,`, `;`, tab, and `|` - Uses encoding fallback (`utf8` then `latin1`) for decoding issues
src/index.js
import fs from "fs";
import path from "path";
function parseArgs(argv) {
const options = { output: "data_profile.json" };
const positional = [];
for (let i = 0; i < argv.length; i += 1) {
const token = argv[i];
if (token.startsWith("--")) {
const key = token.slice(2);
const next = argv[i + 1];
if (next && !next.startsWith("--")) {
options[key] = next;
i += 1;
} else {
options[key] = true;
}
} else {
positional.push(token);
}
}
return { options, positional };
}
function parseCsvLine(line, delimiter) {
const out = [];
let current = "";
let inQuotes = false;
for (let i = 0; i < line.length; i += 1) {
const ch = line[i];
if (ch === '"') {
if (inQuotes && line[i + 1] === '"') {
current += '"';
i += 1;
} else {
inQuotes = !inQuotes;
}
} else if (ch === delimiter && !inQuotes) {
out.push(current);
current = "";
} else {
current += ch;
}
}
out.push(current);
return out;
}
function detectDelimiter(lines) {
const candidates = [",", ";", "\t", "|"];
let best = ",";
let bestScore = -1;
for (const delim of candidates) {
let score = 0;
for (const line of lines) {
score += (line.match(new RegExp(`\\${delim}`, "g")) || []).length;
}
if (score > bestScore) {
bestScore = score;
best = delim;
}
}
return best;
}
function readTextWithFallback(filePath) {
const raw = fs.readFileSync(filePath);
const utf8 = raw.toString("utf8");
const replacementCount = (utf8.match(/\uFFFD/g) || []).length;
if (replacementCount > 0) {
return { text: raw.toString("latin1"), encoding: "latin1" };
}
return { text: utf8, encoding: "utf8" };
}
function loadCsv(filePath) {
const { text, encoding } = readTextWithFallback(filePath);
const lines = text.replace(/\r\n/g, "\n").replace(/\r/g, "\n").split("\n").filter((l) => l.trim().length > 0);
if (lines.length === 0) return { rows: [], encoding, delimiter: "," };
const delimiter = detectDelimiter(lines.slice(0, 5));
const headers = parseCsvLine(lines[0], delimiter).map((h) => h.trim());
const rows = [];
for (const line of lines.slice(1)) {
const values = parseCsvLine(line, delimiter);
const obj = {};
for (let i = 0; i < headers.length; i += 1) {
obj[headers[i]] = i < values.length ? values[i] : "";
}
rows.push(obj);
}
return { rows, encoding, delimiter };
}
function loadJson(filePath) {
const { text, encoding } = readTextWithFallback(filePath);
let data = [];
try {
const parsed = JSON.parse(text);
if (Array.isArray(parsed)) data = parsed;
else if (parsed && typeof parsed === "object") data = [parsed];
} catch {
const lines = text.split(/\r?\n/).map((l) => l.trim()).filter(Boolean);
data = lines.map((line) => JSON.parse(line));
}
return { rows: data.filter((x) => x && typeof x === "object"), encoding };
}
function toNumber(v) {
if (v == null) return null;
const s = String(v).trim();
if (!s) return null;
if (!/^[-+]?\d+(\.\d+)?$/.test(s)) return null;
return Number(s);
}
function toBoolean(v) {
if (v == null) return null;
const s = String(v).trim().toLowerCase();
if (["true", "1", "yes", "y"].includes(s)) return true;
if (["false", "0", "no", "n"].includes(s)) return false;
return null;
}
function toDate(v) {
if (v == null) return null;
const s = String(v).trim();
if (!s) return null;
const ts = Date.parse(s);
return Number.isNaN(ts) ? null : ts;
}
function isMissing(v) {
if (v == null) return true;
const s = String(v).trim().toLowerCase();
return s === "" || s === "null" || s === "na" || s === "n/a" || s === "none";
}
function percentile(sorted, p) {
if (sorted.length === 0) return null;
if (sorted.length === 1) return sorted[0];
const pos = (sorted.length - 1) * p;
const lo = Math.floor(pos);
const hi = Math.ceil(pos);
if (lo === hi) return sorted[lo];
const w = pos - lo;
return sorted[lo] + (sorted[hi] - sorted[lo]) * w;
}
function mean(arr) {
if (arr.length === 0) return null;
return arr.reduce((a, b) => a + b, 0) / arr.length;
}
function stddev(arr, m = null) {
if (arr.length <= 1) return 0;
const mu = m == null ? mean(arr) : m;
const variance = arr.reduce((s, x) => s + (x - mu) ** 2, 0) / arr.length;
return Math.sqrt(variance);
}
function skewness(arr, m = null, sd = null) {
if (arr.length < 3) return 0;
const mu = m == null ? mean(arr) : m;
const sigma = sd == null ? stddev(arr, mu) : sd;
if (sigma === 0) return 0;
const m3 = arr.reduce((s, x) => s + (x - mu) ** 3, 0) / arr.length;
return m3 / sigma ** 3;
}
function topFrequencies(values, n = 10) {
const m = new Map();
for (const v of values) m.set(v, (m.get(v) || 0) + 1);
return [...m.entries()]
.sort((a, b) => (b[1] - a[1]) || String(a[0]).localeCompare(String(b[0])))
.slice(0, n)
.map(([value, count]) => ({ value, count }));
}
function detectColumnType(nonMissingValues) {
if (nonMissingValues.length === 0) return "string";
const numberValues = nonMissingValues.map(toNumber);
const numberCount = numberValues.filter((v) => v != null).length;
const boolValues = nonMissingValues.map(toBoolean);
const boolCount = boolValues.filter((v) => v != null).length;
const dateValues = nonMissingValues.map(toDate);
const dateCount = dateValues.filter((v) => v != null).length;
if (boolCount === nonMissingValues.length) return "boolean";
if (numberCount === nonMissingValues.length) {
const hasFloat = nonMissingValues.some((v) => String(v).includes("."));
return hasFloat ? "float" : "integer";
}
if (dateCount >= Math.max(3, Math.floor(nonMissingValues.length * 0.9))) return "date";
const uniqueCount = new Set(nonMissingValues.map((v) => String(v))).size;
const uniqueRatio = uniqueCount / nonMissingValues.length;
if (uniqueCount <= 20 || uniqueRatio <= 0.1) return "categorical";
return "string";
}
function correlation(xs, ys) {
const paired = [];
const n = Math.min(xs.length, ys.length);
for (let i = 0; i < n; i += 1) {
if (xs[i] == null || ys[i] == null) continue;
paired.push([xs[i], ys[i]]);
}
if (paired.length < 2) return null;
const xvals = paired.map(([x]) => x);
const yvals = paired.map(([, y]) => y);
const mx = mean(xvals);
const my = mean(yvals);
const sx = stddev(xvals, mx);
const sy = stddev(yvals, my);
if (sx === 0 || sy === 0) return 0;
let cov = 0;
for (let i = 0; i < paired.length; i += 1) cov += (paired[i][0] - mx) * (paired[i][1] - my);
cov /= paired.length;
return cov / (sx * sy);
}
function profile(rows) {
const columns = [...new Set(rows.flatMap((row) => Object.keys(row)))];
const columnProfiles = {};
const issues = [];
const numericColumns = [];
const numericSeries = {};
for (const col of columns) {
const raw = rows.map((r) => (Object.prototype.hasOwnProperty.call(r, col) ? r[col] : null));
const missing = raw.filter(isMissing).length;
const total = raw.length;
const missingPct = total === 0 ? 0 : (missing / total) * 100;
const nonMissing = raw.filter((v) => !isMissing(v));
const nonMissingStrings = nonMissing.map((v) => String(v));
const uniqueCount = new Set(nonMissingStrings).size;
const type = detectColumnType(nonMissingStrings);
const profileCol = {
type,
total_values: total,
missing_values: missing,
missing_percentage: Number(missingPct.toFixed(4)),
unique_values: uniqueCount,
quality_issues: [],
};
if (missing === total) {
profileCol.quality_issues.push("entirely_null");
issues.push({ column: col, issue: "entirely_null" });
}
if (uniqueCount === 1 && nonMissingStrings.length > 0) {
profileCol.quality_issues.push("single_unique_value");
issues.push({ column: col, issue: "single_unique_value" });
}
if (type === "integer" || type === "float") {
const nums = nonMissingStrings.map((v) => Number(v)).filter((v) => Number.isFinite(v)).sort((a, b) => a - b);
const mu = mean(nums);
const sd = stddev(nums, mu);
const outliers = nums.filter((v) => sd > 0 && Math.abs(v - mu) > 4 * sd);
if (outliers.length > 0) {
profileCol.quality_issues.push("extreme_outliers");
issues.push({ column: col, issue: "extreme_outliers", count: outliers.length });
}
profileCol.numeric_stats = {
min: nums.length ? nums[0] : null,
max: nums.length ? nums[nums.length - 1] : null,
mean: mu,
median: percentile(nums, 0.5),
stddev: sd,
skewness: skewness(nums, mu, sd),
percentiles: {
p25: percentile(nums, 0.25),
p50: percentile(nums, 0.5),
p75: percentile(nums, 0.75),
p95: percentile(nums, 0.95),
p99: percentile(nums, 0.99),
},
outliers_beyond_4std: outliers,
};
numericColumns.push(col);
numericSeries[col] = raw.map((v) => (isMissing(v) ? null : toNumber(v)));
} else if (type === "string" || type === "categorical") {
const lengths = nonMissingStrings.map((s) => s.length);
const numericLike = nonMissingStrings.filter((s) => toNumber(s) != null).length;
const dateLike = nonMissingStrings.filter((s) => toDate(s) != null).length;
if (nonMissingStrings.length > 0 && numericLike / nonMissingStrings.length >= 0.8) {
profileCol.quality_issues.push("string_looks_numeric");
issues.push({ column: col, issue: "string_looks_numeric" });
}
if (nonMissingStrings.length > 0 && dateLike / nonMissingStrings.length >= 0.8) {
profileCol.quality_issues.push("string_looks_date");
issues.push({ column: col, issue: "string_looks_date" });
}
profileCol.string_stats = {
min_length: lengths.length ? Math.min(...lengths) : null,
max_length: lengths.length ? Math.max(...lengths) : null,
avg_length: lengths.length ? mean(lengths) : null,
unique_count: uniqueCount,
top_values: topFrequencies(nonMissingStrings, 10),
};
}
columnProfiles[col] = profileCol;
}
const correlations = {};
for (const c1 of numericColumns) {
correlations[c1] = {};
for (const c2 of numericColumns) {
correlations[c1][c2] = c1 === c2 ? 1 : correlation(numericSeries[c1], numericSeries[c2]);
}
}
return {
row_count: rows.length,
column_count: columns.length,
columns: columnProfiles,
correlations,
issues,
};
}
function printSummary(report) {
const header = ["Column", "Type", "Total", "Missing%", "Unique", "Notes"];
const rows = Object.entries(report.columns).map(([name, info]) => [
name,
info.type,
String(info.total_values),
info.missing_percentage.toFixed(2),
String(info.unique_values),
info.quality_issues.join(", "),
]);
const widths = header.map((h, i) => Math.max(h.length, ...rows.map((r) => r[i].length)));
const sep = `+-${widths.map((w) => "-".repeat(w)).join("-+-")}-+`;
const fmt = (arr) => `| ${arr.map((v, i) => `${v}${" ".repeat(widths[i] - v.length)}`).join(" | ")} |`;
console.log("Data Profiling Summary");
console.log("======================");
console.log(`Rows: ${report.row_count}`);
console.log(`Columns: ${report.column_count}`);
console.log(sep);
console.log(fmt(header));
console.log(sep);
for (const r of rows) console.log(fmt(r));
console.log(sep);
}
function generateSampleRows() {
const rows = [];
for (let i = 0; i < 1000; i += 1) {
const age = i % 200 === 0 ? 140 : 20 + (i % 45);
const income = i % 150 === 0 ? 250000 : 30000 + (i % 120) * 800 + (i % 7) * 13.5;
const isActive = i % 2 === 0 ? "true" : "false";
const countries = ["US", "CA", "GB", "DE", "IN"];
rows.push({
id: String(i + 1),
age: String(age),
income: income.toFixed(2),
is_active: isActive,
signup_date: `2025-${String((i % 12) + 1).padStart(2, "0")}-${String((i % 28) + 1).padStart(2, "0")}T12:00:00Z`,
country: countries[i % countries.length],
status_code_str: String(1000 + (i % 4)),
comment: i % 25 === 0 ? "" : `note_${i % 17}`,
constant_col: "CONST",
all_null_col: "",
});
}
return rows;
}
function writeSampleCsv(rows, filePath) {
const headers = Object.keys(rows[0]);
const lines = [headers.join(",")];
for (const row of rows) {
lines.push(headers.map((h) => {
const v = row[h] ?? "";
const s = String(v);
if (/[",\n]/.test(s)) return `"${s.replace(/"/g, "\"\"")}"`;
return s;
}).join(","));
}
fs.writeFileSync(filePath, `${lines.join("\n")}\n`, "utf8");
}
function main() {
const { options, positional } = parseArgs(process.argv.slice(2));
const outputPath = path.resolve(options.output || "data_profile.json");
let rows;
let metadata = {};
if (positional.length === 0) {
rows = generateSampleRows();
const samplePath = path.resolve("sample_profile_data.csv");
writeSampleCsv(rows, samplePath);
metadata = { generated_sample: samplePath };
console.log(`No input file provided. Generated sample dataset: ${samplePath}`);
} else {
const inputPath = path.resolve(positional[0]);
if (!fs.existsSync(inputPath)) {
console.error(`Input file not found: ${inputPath}`);
process.exit(1);
}
const ext = path.extname(inputPath).toLowerCase();
if (ext === ".json" || ext === ".jsonl" || ext === ".ndjson") {
const loaded = loadJson(inputPath);
rows = loaded.rows;
metadata = { input_file: inputPath, input_format: "json", encoding: loaded.encoding };
} else {
const loaded = loadCsv(inputPath);
rows = loaded.rows;
metadata = { input_file: inputPath, input_format: "csv", encoding: loaded.encoding, delimiter: loaded.delimiter };
}
}
const reportBody = profile(rows);
const report = {
generated_at: new Date().toISOString(),
...metadata,
...reportBody,
};
printSummary(report);
fs.writeFileSync(outputPath, `${JSON.stringify(report, null, 2)}\n`, "utf8");
console.log(`Saved JSON profile: ${outputPath}`);
}
main();