CSV Statistical Analyzer (javascript, written by Claude Code)
envgap__claude-code__javascript-t1-1
Written by a coding agent; not on GitHubWritten 2026-02-27
01 / FAILURE SIGNATURE
As the study recorded it
No identifying execution failure has been captured.
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
package.json- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
claude-code/javascript-t1 #1 · read the task the agent was given
Claude Code wrote this javascript project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written. Task given to the agent: TASK: CSV Statistical Analyzer Write a program that reads a CSV file and performs comprehensive statistical analysis on every numeric column. It should handle real-world messy data — missing values, mixed types, malformed rows — and produce both a human-readable console report and a machine-readable JSON output. FUNCTIONAL REQUIREMENTS: - Accept a CSV file path as a command-line argument - Auto-detect which columns are numeric vs categorical - For each numeric column compute: mean, median, standard deviation, variance, min, max, 25th/50th/75th percentiles, and non-missing value count - Detect outliers using the IQR method (values below Q1 - 1.5*IQR or above Q3 + 1.5*IQR) and list them per column - For each categorical column compute: unique count, most frequent value, and top 10 value frequencies - Print a formatted summary table to the console with aligned columns - Save the complete analysis to report.json including all stats, outlier details, and column type classifications - If no input file is given, generate a sample CSV with at least 200 rows across 5 numeric and 2 categorical columns, then analyze it - Handle gracefully: empty files, header-only files, columns with all missing values, single-row files, quoted fields containing commas Create a complete JavaScript project for a clean Ubuntu 22.04 machine with only Node.js 20+ (LTS) installed. Include: - Source code - package.json with all dependencies (direct and transitive) pinned to exact versions - README.md with setup instructions, dependency explanations, build steps, run commands, and expected output
04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
4 files, exactly as written, before any repair.
generation_metadata.json
{
"project": "p_01",
"language": "javascript",
"trial": 1,
"agent": "claude",
"model": "sonnet",
"elapsed_seconds": 244.94,
"exit_code": 0,
"files_created": [
"index.js",
"package.json",
"README.md"
],
"timestamp": "2026-02-27T04:44:19Z",
"prompt_file": "javascript_prompts/p_01.md"
}index.js
#!/usr/bin/env node
'use strict';
const fs = require('fs');
const path = require('path');
const { parse } = require('csv-parse/sync');
const Table = require('cli-table3');
// ─── Statistical helpers ──────────────────────────────────────────────────────
function computeMean(nums) {
if (!nums.length) return null;
return nums.reduce((s, v) => s + v, 0) / nums.length;
}
function computeVariance(nums) {
if (nums.length < 2) return null;
const m = computeMean(nums);
return nums.reduce((s, v) => s + (v - m) ** 2, 0) / (nums.length - 1);
}
function computeStdDev(nums) {
const v = computeVariance(nums);
return v === null ? null : Math.sqrt(v);
}
function computePercentile(sorted, p) {
if (!sorted.length) return null;
const idx = (p / 100) * (sorted.length - 1);
const lo = Math.floor(idx);
const hi = Math.ceil(idx);
if (lo === hi) return sorted[lo];
return sorted[lo] + (sorted[hi] - sorted[lo]) * (idx - lo);
}
function computeNumericStats(values) {
const nums = values.filter(v => v !== null && !isNaN(v));
if (!nums.length) {
return {
count: 0,
mean: null, median: null, stddev: null, variance: null,
min: null, max: null, p25: null, p50: null, p75: null,
outlierCount: 0, outliers: [],
};
}
const sorted = [...nums].sort((a, b) => a - b);
const mean = computeMean(nums);
const variance = computeVariance(nums);
const q1 = computePercentile(sorted, 25);
const q3 = computePercentile(sorted, 75);
const iqr = q3 - q1;
const lowerFence = q1 - 1.5 * iqr;
const upperFence = q3 + 1.5 * iqr;
const outliers = nums.filter(v => v < lowerFence || v > upperFence);
return {
count: nums.length,
mean,
median: computePercentile(sorted, 50),
stddev: variance !== null ? Math.sqrt(variance) : null,
variance,
min: sorted[0],
max: sorted[sorted.length - 1],
p25: q1,
p50: computePercentile(sorted, 50),
p75: q3,
lowerFence,
upperFence,
outlierCount: outliers.length,
outliers,
};
}
function computeCategoricalStats(values) {
const nonNull = values.filter(v => v !== null && v !== '');
const freq = {};
for (const v of nonNull) freq[v] = (freq[v] || 0) + 1;
const sorted = Object.entries(freq).sort((a, b) => b[1] - a[1]);
const top10 = sorted.slice(0, 10);
return {
count: nonNull.length,
missingCount: values.length - nonNull.length,
uniqueCount: Object.keys(freq).length,
mostFrequent: sorted.length ? sorted[0][0] : null,
mostFrequentCount: sorted.length ? sorted[0][1] : 0,
top10Frequencies: Object.fromEntries(top10),
};
}
// ─── Column type detection ────────────────────────────────────────────────────
function isNumericColumn(values) {
const nonEmpty = values.filter(v => v !== null && v !== '' && v !== undefined);
if (!nonEmpty.length) return false;
const numericCount = nonEmpty.filter(v => !isNaN(parseFloat(v)) && isFinite(Number(v))).length;
return numericCount / nonEmpty.length >= 0.8;
}
// ─── CSV parsing ──────────────────────────────────────────────────────────────
function parseCSV(filePath) {
let content;
try {
content = fs.readFileSync(filePath, 'utf8');
} catch (err) {
console.error(`Error reading file: ${err.message}`);
process.exit(1);
}
// Strip BOM if present
if (content.charCodeAt(0) === 0xFEFF) content = content.slice(1);
content = content.trim();
if (!content) return { headers: [], rows: [] };
let records;
try {
records = parse(content, {
skip_empty_lines: true,
relax_column_count: true,
trim: true,
});
} catch (err) {
console.error(`Error parsing CSV: ${err.message}`);
process.exit(1);
}
if (!records.length) return { headers: [], rows: [] };
const headers = records[0].map(h => h || '(empty)');
const rows = records.slice(1);
return { headers, rows };
}
// ─── Sample CSV generation ────────────────────────────────────────────────────
function escapeCSVValue(val) {
const str = String(val);
if (str.includes(',') || str.includes('"') || str.includes('\n')) {
return `"${str.replace(/"/g, '""')}"`;
}
return str;
}
function randBetween(lo, hi) {
return lo + Math.random() * (hi - lo);
}
function generateSampleCSV(outputPath) {
const categories = ['Electronics', 'Clothing', 'Food', 'Books', 'Sports'];
// Cities intentionally include one with a comma to exercise quoted-field parsing
const cities = ['New York', 'Los Angeles', 'Chicago', 'Houston', 'Phoenix', 'Portland, OR'];
const headers = ['age', 'salary', 'score', 'temperature', 'price', 'category', 'city'];
const lines = [headers.join(',')];
for (let i = 0; i < 210; i++) {
let age = Math.floor(randBetween(20, 80));
let salary = Math.floor(randBetween(30000, 130000));
let score = randBetween(0, 100).toFixed(2);
let temperature = randBetween(-10, 40).toFixed(1);
let price = randBetween(10, 500).toFixed(2);
const category = categories[Math.floor(Math.random() * categories.length)];
const city = cities[Math.floor(Math.random() * cities.length)];
// Inject high outliers periodically
if (i % 42 === 0) salary = Math.floor(randBetween(250000, 650000));
if (i % 55 === 0) age = Math.floor(randBetween(90, 115));
if (i % 70 === 0) price = randBetween(2000, 7000).toFixed(2);
const row = [age, salary, score, temperature, price, category, city];
// Inject ~5% missing values in numeric columns
if (Math.random() < 0.05) {
const idx = Math.floor(Math.random() * 5);
row[idx] = '';
}
lines.push(row.map(escapeCSVValue).join(','));
}
fs.writeFileSync(outputPath, lines.join('\n'), 'utf8');
console.log(`Generated sample CSV: ${outputPath} (${lines.length - 1} data rows)\n`);
return outputPath;
}
// ─── Formatting helpers ───────────────────────────────────────────────────────
function fmt(v, digits = 4) {
if (v === null || v === undefined) return 'N/A';
if (typeof v === 'number') {
if (!isFinite(v)) return 'N/A';
return Number.isInteger(v) ? String(v) : v.toFixed(digits);
}
return String(v);
}
// ─── Main analysis ────────────────────────────────────────────────────────────
function analyzeCSV(filePath) {
console.log(`Analyzing: ${path.resolve(filePath)}\n`);
const { headers, rows } = parseCSV(filePath);
if (!headers.length) {
console.error('Error: File is empty or contains no parseable data.');
process.exit(1);
}
if (!rows.length) {
console.log('Warning: File contains a header row only — no data to analyze.');
const report = {
file: path.resolve(filePath),
generatedAt: new Date().toISOString(),
totalRows: 0,
totalColumns: headers.length,
columns: {},
};
fs.writeFileSync('report.json', JSON.stringify(report, null, 2), 'utf8');
console.log('Saved report.json');
return;
}
// Collect per-column raw string values
const columnValues = {};
for (const h of headers) columnValues[h] = [];
for (const row of rows) {
for (let i = 0; i < headers.length; i++) {
const raw = row[i] !== undefined ? String(row[i]).trim() : '';
columnValues[headers[i]].push(raw === '' ? null : raw);
}
}
// Classify columns and compute stats
const numericColumns = {};
const categoricalColumns = {};
for (const h of headers) {
const values = columnValues[h];
if (isNumericColumn(values)) {
const nums = values.map(v => {
if (v === null) return null;
const n = parseFloat(v);
return isFinite(n) ? n : null;
});
numericColumns[h] = computeNumericStats(nums);
} else {
categoricalColumns[h] = computeCategoricalStats(values);
}
}
const totalRows = rows.length;
const numColCount = Object.keys(numericColumns).length;
const catColCount = Object.keys(categoricalColumns).length;
console.log(
`Rows: ${totalRows} | Columns: ${headers.length} ` +
`(${numColCount} numeric, ${catColCount} categorical)\n`
);
// ─── Numeric summary table ──────────────────────────────────────────────────
if (numColCount > 0) {
console.log('NUMERIC COLUMNS');
console.log('─'.repeat(80));
const numTable = new Table({
head: ['Column', 'Count', 'Mean', 'Median', 'Std Dev', 'Min', 'Max', 'P25', 'P75', 'Outliers'],
style: { head: ['cyan'], border: [] },
});
for (const [col, s] of Object.entries(numericColumns)) {
numTable.push([
col,
s.count,
fmt(s.mean),
fmt(s.median),
fmt(s.stddev),
fmt(s.min),
fmt(s.max),
fmt(s.p25),
fmt(s.p75),
s.outlierCount,
]);
}
console.log(numTable.toString());
console.log();
}
// ─── Categorical summary table ──────────────────────────────────────────────
if (catColCount > 0) {
console.log('CATEGORICAL COLUMNS');
console.log('─'.repeat(80));
const catTable = new Table({
head: ['Column', 'Count', 'Missing', 'Unique', 'Most Frequent', 'Freq Count'],
style: { head: ['cyan'], border: [] },
});
for (const [col, s] of Object.entries(categoricalColumns)) {
catTable.push([
col,
s.count,
s.missingCount,
s.uniqueCount,
s.mostFrequent !== null ? s.mostFrequent : 'N/A',
s.mostFrequentCount,
]);
}
console.log(catTable.toString());
console.log();
// Per-column top-10 frequency breakdown
for (const [col, s] of Object.entries(categoricalColumns)) {
console.log(` "${col}" — value frequencies (top ${Object.keys(s.top10Frequencies).length}):`);
for (const [val, count] of Object.entries(s.top10Frequencies)) {
const pct = s.count > 0 ? ((count / s.count) * 100).toFixed(1) : '0.0';
const bar = '█'.repeat(Math.round((count / s.count) * 30));
console.log(` ${val.padEnd(22)} ${String(count).padStart(5)} (${pct.padStart(5)}%) ${bar}`);
}
console.log();
}
}
// ─── Outlier details ────────────────────────────────────────────────────────
const colsWithOutliers = Object.entries(numericColumns).filter(([, s]) => s.outlierCount > 0);
if (colsWithOutliers.length) {
console.log('OUTLIERS (IQR method: outside Q1 − 1.5×IQR … Q3 + 1.5×IQR)');
console.log('─'.repeat(80));
for (const [col, s] of colsWithOutliers) {
const fences = `fence [${fmt(s.lowerFence, 2)}, ${fmt(s.upperFence, 2)}]`;
const preview = s.outliers.slice(0, 10).map(v => fmt(v, 2)).join(', ');
const more = s.outliers.length > 10 ? ` … +${s.outliers.length - 10} more` : '';
console.log(` ${col.padEnd(15)} ${s.outlierCount} outlier(s) ${fences}`);
console.log(` values: [${preview}${more}]`);
}
console.log();
}
// ─── Save report.json ───────────────────────────────────────────────────────
const report = {
file: path.resolve(filePath),
generatedAt: new Date().toISOString(),
totalRows,
totalColumns: headers.length,
columnOrder: headers,
columns: {},
};
for (const [col, s] of Object.entries(numericColumns)) {
report.columns[col] = { type: 'numeric', ...s };
}
for (const [col, s] of Object.entries(categoricalColumns)) {
report.columns[col] = { type: 'categorical', ...s };
}
fs.writeFileSync('report.json', JSON.stringify(report, null, 2), 'utf8');
console.log('Saved: report.json');
}
// ─── Entry point ──────────────────────────────────────────────────────────────
function main() {
let filePath = process.argv[2];
if (!filePath) {
filePath = 'sample_data.csv';
generateSampleCSV(filePath);
} else if (!fs.existsSync(filePath)) {
console.error(`Error: File not found: ${filePath}`);
process.exit(1);
}
analyzeCSV(filePath);
}
main();
package.json
{
"name": "csv-statistical-analyzer",
"version": "1.0.0",
"description": "Comprehensive statistical analysis for CSV files — numeric stats, percentiles, IQR outlier detection, and categorical frequency tables.",
"main": "index.js",
"scripts": {
"start": "node index.js",
"analyze": "node index.js"
},
"dependencies": {
"cli-table3": "0.6.4",
"csv-parse": "5.5.5"
},
"engines": {
"node": ">=20.0.0"
},
"license": "MIT"
}
README.md
# CSV Statistical Analyzer
A Node.js command-line tool that reads a CSV file and performs comprehensive statistical analysis on every column. It handles real-world messy data (missing values, mixed types, malformed rows, quoted fields containing commas) and produces both a human-readable console report and a machine-readable `report.json` file.
---
## Prerequisites
- **Node.js 20 LTS** or later — https://nodejs.org/en/download
Verify: `node --version` (should print `v20.x.x` or higher)
- **npm** (bundled with Node.js)
No other system software is required.
---
## Installation
```bash
# 1. Navigate to the project directory
cd csv-statistical-analyzer
# 2. Install dependencies
npm install
```
`npm install` reads `package.json`, downloads the two direct dependencies (and their transitive dependencies) from the npm registry, and writes a `package-lock.json` that pins every transitive package to an exact version for reproducible installs.
---
## Running the Program
### Analyze your own CSV file
```bash
node index.js path/to/your/data.csv
```
### Auto-generate a sample dataset and analyze it
When no file path is supplied the program generates `sample_data.csv` (210 rows × 7 columns) and immediately analyzes it:
```bash
node index.js
# or
npm start
```
### npm script shorthand
```bash
npm run analyze -- path/to/your/data.csv
```
---
## Expected Output
```
Analyzing: /home/user/project/sample_data.csv
Rows: 210 | Columns: 7 (5 numeric, 2 categorical)
NUMERIC COLUMNS
────────────────────────────────────────────────────────────────────────────────
┌─────────────┬───────┬──────────┬──────────┬─────────┬──────────┬───────────┬──────────┬──────────┬─────────┐
│ Column │ Count │ Mean │ Median │ Std Dev │ Min │ Max │ P25 │ P75 │Outliers │
├─────────────┼───────┼──────────┼──────────┼─────────┼──────────┼───────────┼──────────┼──────────┼─────────┤
│ age │ 207 │ 49.4638 │ 49.0000 │ 19.2891 │ 20.0000 │ 109.0000 │ 34.0000 │ 65.0000 │ 5 │
│ salary │ 208 │ 93714.73 │ 79864.00 │ 84561.2 │ 30012.00 │ 617343.00 │ 52031.75 │ 107853.5 │ 5 │
...
└─────────────┴───────┴──────────┴──────────┴─────────┴──────────┴───────────┴──────────┴──────────┴─────────┘
CATEGORICAL COLUMNS
────────────────────────────────────────────────────────────────────────────────
┌──────────┬───────┬─────────┬────────┬───────────────┬────────────┐
│ Column │ Count │ Missing │ Unique │ Most Frequent │ Freq Count │
├──────────┼───────┼─────────┼────────┼───────────────┼────────────┤
│ category │ 210 │ 0 │ 5 │ Food │ 48 │
│ city │ 210 │ 0 │ 6 │ Chicago │ 41 │
└──────────┴───────┴─────────┴────────┴───────────────┴────────────┘
"category" — value frequencies (top 5):
Food 48 (22.9%) ██████▉
...
OUTLIERS (IQR method: outside Q1 − 1.5×IQR … Q3 + 1.5×IQR)
────────────────────────────────────────────────────────────────────────────────
age 5 outlier(s) fence [−13.50, 112.50]
values: [95.00, 102.00, ...]
...
Saved: report.json
```
A `report.json` file is written to the current working directory containing all statistics, outlier value lists, and column type classifications.
---
## Output: report.json structure
```jsonc
{
"file": "/absolute/path/to/input.csv",
"generatedAt": "2024-01-15T12:00:00.000Z",
"totalRows": 210,
"totalColumns": 7,
"columnOrder": ["age", "salary", ...],
"columns": {
"age": {
"type": "numeric",
"count": 207, // non-missing values
"mean": 49.46,
"median": 49.0,
"stddev": 19.29,
"variance": 372.15,
"min": 20,
"max": 109,
"p25": 34.0,
"p50": 49.0,
"p75": 65.0,
"lowerFence": -13.5, // Q1 − 1.5 × IQR
"upperFence": 112.5, // Q3 + 1.5 × IQR
"outlierCount": 5,
"outliers": [95, 102, ...]
},
"category": {
"type": "categorical",
"count": 210,
"missingCount": 0,
"uniqueCount": 5,
"mostFrequent": "Food",
"mostFrequentCount": 48,
"top10Frequencies": { "Food": 48, "Books": 45, ... }
}
}
}
```
---
## Edge Cases Handled
| Situation | Behavior |
|-----------|----------|
| Empty file | Exits with a clear error message |
| Header-only file | Reports 0 rows; writes minimal `report.json` |
| Column with all missing values | Classified as categorical; all stats are `null` / `0` |
| Single data row | Stats computed where possible; variance/stddev shown as `null` |
| Quoted fields containing commas | Parsed correctly by `csv-parse` (e.g. `"Portland, OR"`) |
| Rows with fewer columns than header | Missing cells treated as empty/null |
| Mixed-type column (mostly numeric) | If ≥ 80 % of non-empty values are numeric, treated as numeric |
| UTF-8 BOM | Stripped automatically before parsing |
---
## Dependencies
### Direct dependencies
| Package | Version | Purpose |
|---------|---------|---------|
| [`csv-parse`](https://csv.js.org/parse/) | `5.5.5` | RFC-4180-compliant CSV parser. Handles quoted fields, multi-line values, relaxed column counts, and BOM stripping. |
| [`cli-table3`](https://github.com/cli-table/cli-table3) | `0.6.4` | Renders bordered, column-aligned tables in the terminal with optional ANSI colour for headers. |
### Transitive dependencies (installed automatically by npm)
| Package | Pulled in by | Purpose |
|---------|-------------|---------|
| `@colors/colors@1.5.0` | cli-table3 | ANSI terminal colour codes for table header styling |
| `string-width@4.2.3` | cli-table3 | Measures the visible width of a string (handles Unicode / CJK characters) |
| `emoji-regex@8.0.0` | string-width | Regex to identify emoji code-points when computing display width |
| `is-fullwidth-code-point@3.0.0` | string-width | Detects double-width Unicode characters (e.g. CJK) |
| `strip-ansi@6.0.1` | string-width | Removes ANSI escape codes before measuring a string's length |
| `ansi-regex@5.0.1` | strip-ansi | Regex pattern used to detect ANSI escape sequences |
All statistics (mean, variance, standard deviation, percentiles, IQR) are computed from scratch using pure JavaScript — no external math library is required.
---
## Build steps
There is no compilation or transpilation step. The project runs directly with Node.js:
```bash
npm install # install dependencies once
node index.js # run
```
To verify the installation is healthy:
```bash
node -e "require('csv-parse/sync'); require('cli-table3'); console.log('OK')"
# prints: OK
```
---
## Project structure
```
.
├── index.js # All application logic (single-file)
├── package.json # Manifest with exact dependency versions
├── README.md # This file
├── package-lock.json # Generated by npm install (full transitive lock)
├── sample_data.csv # Generated when no argument is supplied
└── report.json # Written after each analysis run
```