← All tasks
javascriptclaude-code/javascript-t1 #7Not a task: already works

Data Profiling Tool (javascript, written by Claude Code)

envgap__claude-code__javascript-t1-7

Written by a coding agent; not on GitHubWritten 2026-02-27

01 / FAILURE SIGNATURE

As the study recorded it

No identifying execution failure has been captured.
Not a benchmark task.
  • The project already builds and runs before the fix, so there is nothing to repair.

02 / ENVIRONMENT RECIPE

Base commit
Not freshly verified
Manifest
package.json
Reproduce
Awaiting issue-specific recipe
Run under trace
Awaiting a meaningful runtime command

03 / TASK AND FAILURE

claude-code/javascript-t1 #7 · read the task the agent was given
Claude Code wrote this javascript project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written.

Task given to the agent:

TASK: Data Profiling Tool

Write a program that profiles tabular datasets by analyzing column types, distributions, missing values, correlations, and data quality issues, producing a comprehensive data quality report.

FUNCTIONAL REQUIREMENTS:
- Accept a CSV or JSON data file path as a command-line argument
- Auto-detect column data types: numeric (integer vs float), string, boolean, date/timestamp, and categorical (low cardinality strings)
- For numeric columns: compute min, max, mean, median, standard deviation, skewness, and percentiles (25th, 50th, 75th, 95th, 99th)
- For string columns: compute min/max/average length, most common values (top 10), and unique count
- For all columns: count total values, missing/null values, missing percentage, and unique value count
- Detect potential data quality issues: columns that are entirely null, columns with a single unique value, numeric columns with extreme outliers (beyond 4 standard deviations), and string columns that look like misclassified numbers or dates
- Compute a pairwise correlation matrix for all numeric columns
- Print a formatted summary report to console showing key statistics per column
- Save the full profiling report as a JSON file with --output flag (default: data_profile.json)
- If no input file is given, generate a sample dataset with 1000 rows across at least 8 columns of mixed types including intentional quality issues, then profile it
- Handle files with inconsistent delimiters or encoding issues gracefully

Create a complete JavaScript project for a clean Ubuntu 22.04 machine with only Node.js 20+ (LTS) installed. Include:
- Source code
- package.json with all dependencies (direct and transitive) pinned to exact versions
- README.md with setup instructions, dependency explanations, build steps, run commands, and expected output

04 / LABELS

Labels from the report text only; not yet run

No supported category has been assigned.

Label rules and the text that matched
[]

05 / FILES

The project as the agent wrote it

3 files, exactly as written, before any repair.

package.json
{
  "name": "data-profiler-trial1",
  "version": "1.0.0",
  "description": "Data Profiling Tool - Trial 1 (csv-parse + simple-statistics + cli-table3)",
  "main": "profiler.js",
  "scripts": {
    "start": "node profiler.js",
    "profile": "node profiler.js"
  },
  "dependencies": {
    "csv-parse": "5.5.6",
    "simple-statistics": "7.8.3",
    "cli-table3": "0.6.5"
  }
}
profiler.js
#!/usr/bin/env node
/**
 * Data Profiling Tool - Trial 1
 * Dependencies: csv-parse, simple-statistics, cli-table3
 * Profiles tabular datasets: column types, distributions, missing values,
 * correlations, data quality report.
 */

const fs = require("fs");
const path = require("path");
const { parse } = require("csv-parse/sync");
const ss = require("simple-statistics");
const Table = require("cli-table3");

// --- Sample Data Generation ---
function generateSampleDataset(filePath) {
  const rows = 200;
  const names = ["Alice", "Bob", "Charlie", "Diana", "Eve", "", "Frank", "Grace", null, "Heidi"];
  const departments = ["Engineering", "Sales", "HR", "Marketing", null, "Engineering", "Sales", ""];
  const dates = ["2020-01-15", "2021-06-30", "2019-12-01", null, "invalid-date", "2022-03-10"];
  const actives = ["true", "false", null, "yes", "no", "1", "0"];

  let csv = "id,name,age,salary,department,score,join_date,is_active\n";
  for (let i = 0; i < rows; i++) {
    const pick = (arr) => arr[Math.floor(Math.random() * arr.length)];
    const age = pick([Math.floor(Math.random() * 62) + 18, null, -1, 999]);
    const salary = pick([Math.round(Math.random() * 125000 + 25000), null, 0, -500]);
    const score = pick([Math.round(Math.random() * 1000) / 10, null, ""]);
    csv += `${i + 1},${pick(names) || ""},${age || ""},${salary || ""},${pick(departments) || ""},${score === null ? "" : score},${pick(dates) || ""},${pick(actives) || ""}\n`;
  }

  fs.writeFileSync(filePath, csv, "utf-8");
  console.log(`[INFO] Generated sample dataset: ${filePath} (${rows} rows)`);
  return filePath;
}

// --- Data Loading ---
function loadData(filePath) {
  const ext = path.extname(filePath).toLowerCase();
  let records;

  if (ext === ".csv") {
    const content = fs.readFileSync(filePath, "utf-8");
    records = parse(content, {
      columns: true,
      skip_empty_lines: true,
      trim: true,
      cast: false,
    });
  } else if (ext === ".json") {
    const content = fs.readFileSync(filePath, "utf-8");
    records = JSON.parse(content);
    if (!Array.isArray(records)) {
      throw new Error("JSON file must contain an array of objects");
    }
  } else {
    throw new Error(`Unsupported format: ${ext}. Use .csv or .json`);
  }

  return records;
}

// --- Null detection ---
const NULL_VALUES = new Set(["", "null", "NULL", "None", "NA", "N/A", "nan", "NaN", "undefined"]);

function isNull(val) {
  if (val === null || val === undefined) return true;
  if (typeof val === "string" && NULL_VALUES.has(val.trim())) return true;
  return false;
}

function cleanValues(values) {
  return values.filter((v) => !isNull(v));
}

// --- Type Detection ---
function detectColumnType(values) {
  const clean = cleanValues(values);
  if (clean.length === 0) return "empty";

  // Check numeric
  let numericCount = 0;
  for (const v of clean) {
    const n = Number(v);
    if (!isNaN(n) && String(v).trim() !== "") numericCount++;
  }

  if (numericCount / clean.length > 0.8) {
    const allInts = clean.every((v) => {
      const n = Number(v);
      return !isNaN(n) && Number.isInteger(n);
    });
    if (numericCount === clean.length && allInts) return "integer";
    if (numericCount === clean.length) return "float";
    return "numeric_mixed";
  }

  // Check boolean
  const boolVals = new Set(["true", "false", "yes", "no", "1", "0", "y", "n"]);
  const allBool = clean.every((v) => boolVals.has(String(v).toLowerCase().trim()));
  if (allBool) return "boolean";

  // Check datetime
  let dateCount = 0;
  for (const v of clean) {
    const d = new Date(v);
    if (!isNaN(d.getTime()) && String(v).length > 4) dateCount++;
  }
  if (dateCount / clean.length > 0.7) return "datetime";

  // Categorical vs text
  const unique = new Set(clean.map((v) => String(v)));
  if (unique.size / clean.length < 0.5 || unique.size <= 20) return "categorical";
  return "text";
}

// --- Numeric Stats ---
function computeNumericStats(values) {
  const nums = cleanValues(values)
    .map(Number)
    .filter((n) => !isNaN(n));
  if (nums.length === 0) return {};

  return {
    count: nums.length,
    mean: round(ss.mean(nums)),
    median: round(ss.median(nums)),
    std: round(nums.length > 1 ? ss.standardDeviation(nums) : 0),
    min: round(ss.min(nums)),
    max: round(ss.max(nums)),
    q1: round(ss.quantile(nums, 0.25)),
    q3: round(ss.quantile(nums, 0.75)),
    skewness: round(nums.length >= 3 ? ss.sampleSkewness(nums) : 0),
    kurtosis: round(nums.length >= 4 ? ss.sampleKurtosis(nums) : 0),
    zeros: nums.filter((n) => n === 0).length,
    negatives: nums.filter((n) => n < 0).length,
  };
}

// --- Categorical Stats ---
function computeCategoricalStats(values) {
  const clean = cleanValues(values).map(String).filter((v) => v.trim() !== "");
  if (clean.length === 0) return {};

  const freq = {};
  for (const v of clean) {
    freq[v] = (freq[v] || 0) + 1;
  }

  const sorted = Object.entries(freq).sort((a, b) => b[1] - a[1]);
  const topValues = {};
  for (const [k, v] of sorted.slice(0, 10)) {
    topValues[k] = v;
  }

  return {
    unique_count: new Set(clean).size,
    top_values: topValues,
    mode: sorted.length > 0 ? sorted[0][0] : null,
    mode_frequency: sorted.length > 0 ? sorted[0][1] : 0,
  };
}

// --- Distributions ---
function computeDistribution(values, colType) {
  if (["integer", "float", "numeric_mixed"].includes(colType)) {
    const nums = cleanValues(values)
      .map(Number)
      .filter((n) => !isNaN(n));
    if (nums.length === 0) return {};

    const minVal = ss.min(nums);
    const maxVal = ss.max(nums);
    const binCount = 10;
    const binWidth = (maxVal - minVal) / binCount || 1;
    const bins = [];
    const counts = new Array(binCount).fill(0);

    for (let i = 0; i <= binCount; i++) {
      bins.push(round(minVal + i * binWidth));
    }

    for (const n of nums) {
      let idx = Math.floor((n - minVal) / binWidth);
      if (idx >= binCount) idx = binCount - 1;
      if (idx < 0) idx = 0;
      counts[idx]++;
    }

    return { type: "histogram", bins, counts };
  }

  if (["categorical", "boolean", "text"].includes(colType)) {
    const clean = cleanValues(values).map(String);
    const freq = {};
    for (const v of clean) {
      freq[v] = (freq[v] || 0) + 1;
    }
    const sorted = Object.entries(freq)
      .sort((a, b) => b[1] - a[1])
      .slice(0, 20);
    const result = {};
    for (const [k, v] of sorted) result[k] = v;
    return { type: "frequency", values: result };
  }

  return {};
}

// --- Correlation Matrix ---
function computeCorrelationMatrix(data, columns, colTypes) {
  const numericCols = columns.filter((c) =>
    ["integer", "float", "numeric_mixed"].includes(colTypes[c])
  );

  if (numericCols.length < 2) return {};

  const result = {};
  for (const colA of numericCols) {
    result[colA] = {};
    for (const colB of numericCols) {
      // Align: only use rows where both are numeric
      const pairs = [];
      for (const row of data) {
        const a = Number(row[colA]);
        const b = Number(row[colB]);
        if (!isNaN(a) && !isNull(row[colA]) && !isNaN(b) && !isNull(row[colB])) {
          pairs.push([a, b]);
        }
      }

      if (pairs.length < 2) {
        result[colA][colB] = null;
        continue;
      }

      const xs = pairs.map((p) => p[0]);
      const ys = pairs.map((p) => p[1]);
      const corr = ss.sampleCorrelation(xs, ys);
      result[colA][colB] = isNaN(corr) ? null : round(corr);
    }
  }

  return result;
}

// --- Data Quality Score ---
function computeDataQualityScore(data, columns, colTypes) {
  const totalCells = data.length * columns.length;
  if (totalCells === 0) return 0;

  // Completeness
  let nullCount = 0;
  for (const row of data) {
    for (const col of columns) {
      if (isNull(row[col])) nullCount++;
    }
  }
  const completeness = 1 - nullCount / totalCells;

  // Uniqueness
  const uniquenessScores = [];
  for (const col of columns) {
    const clean = cleanValues(data.map((r) => r[col]));
    if (clean.length > 0) {
      uniquenessScores.push(new Set(clean.map(String)).size / clean.length);
    }
  }
  const avgUniqueness =
    uniquenessScores.length > 0
      ? uniquenessScores.reduce((a, b) => a + b, 0) / uniquenessScores.length
      : 0;

  // Consistency
  const consistencyScores = [];
  for (const col of columns) {
    const clean = cleanValues(data.map((r) => r[col]));
    if (clean.length > 0) {
      const numCount = clean.filter((v) => !isNaN(Number(v)) && String(v).trim() !== "").length;
      const ratio = numCount / clean.length;
      consistencyScores.push(Math.max(ratio, 1 - ratio));
    }
  }
  const avgConsistency =
    consistencyScores.length > 0
      ? consistencyScores.reduce((a, b) => a + b, 0) / consistencyScores.length
      : 0;

  const score = (completeness * 0.5 + avgConsistency * 0.3 + Math.min(avgUniqueness, 1) * 0.2) * 100;
  return round(score);
}

// --- Utility ---
function round(val, decimals = 4) {
  if (val === null || val === undefined || isNaN(val)) return null;
  return Math.round(val * Math.pow(10, decimals)) / Math.pow(10, decimals);
}

// --- Console Report ---
function printConsoleReport(report) {
  console.log(`\n${"=".repeat(60)}`);
  console.log("  DATA PROFILING TOOL - Trial 1 (csv-parse + simple-statistics + cli-table3)");
  console.log(`${"=".repeat(60)}\n`);

  // Overview table
  const overviewTable = new Table({
    head: ["Metric", "Value"],
    colWidths: [25, 35],
  });
  const ov = report.dataset_overview;
  overviewTable.push(
    ["Rows", ov.rows],
    ["Columns", ov.columns],
    ["Total Cells", ov.total_cells],
    ["Total Missing", `${ov.total_missing} (${ov.total_missing_pct}%)`],
    ["Quality Score", `${report.data_quality_score} / 100`]
  );
  console.log(overviewTable.toString());
  console.log();

  // Column profiles table
  const colTable = new Table({
    head: ["Column", "Type", "Missing", "Miss%", "Unique", "Min", "Max", "Mean", "Median", "Mode"],
    colWidths: [14, 14, 9, 8, 8, 10, 10, 10, 10, 12],
  });

  for (const [colName, profile] of Object.entries(report.column_profiles)) {
    const row = [
      colName,
      profile.detected_type,
      profile.missing_count,
      `${profile.missing_percentage}%`,
      profile.unique_count,
    ];

    if (profile.numeric_stats && Object.keys(profile.numeric_stats).length > 0) {
      const ns = profile.numeric_stats;
      row.push(ns.min ?? "", ns.max ?? "", ns.mean ?? "", ns.median ?? "", "");
    } else if (profile.categorical_stats && Object.keys(profile.categorical_stats).length > 0) {
      row.push("", "", "", "", profile.categorical_stats.mode || "");
    } else {
      row.push("", "", "", "", "");
    }

    colTable.push(row);
  }
  console.log(colTable.toString());
  console.log();

  // Detailed per-column stats
  for (const [colName, profile] of Object.entries(report.column_profiles)) {
    if (profile.numeric_stats && Object.keys(profile.numeric_stats).length > 0) {
      console.log(`  --- ${colName} (Numeric) ---`);
      const ns = profile.numeric_stats;
      for (const [k, v] of Object.entries(ns)) {
        console.log(`    ${k.padEnd(15)} ${v}`);
      }
    }
    if (profile.categorical_stats && Object.keys(profile.categorical_stats).length > 0) {
      console.log(`  --- ${colName} (Categorical) ---`);
      const cs = profile.categorical_stats;
      console.log(`    Unique: ${cs.unique_count}`);
      console.log(`    Mode: ${cs.mode} (freq: ${cs.mode_frequency})`);
      console.log(`    Top values:`);
      for (const [val, cnt] of Object.entries(cs.top_values).slice(0, 5)) {
        console.log(`      ${val}: ${cnt}`);
      }
    }
  }

  // Correlation matrix
  if (report.correlation_matrix && Object.keys(report.correlation_matrix).length > 0) {
    console.log("\n--- Correlation Matrix ---");
    const cols = Object.keys(report.correlation_matrix);
    const corrTable = new Table({
      head: ["", ...cols],
    });
    for (const row of cols) {
      const r = [row];
      for (const col of cols) {
        const val = report.correlation_matrix[col]?.[row];
        r.push(val !== null && val !== undefined ? val : "N/A");
      }
      corrTable.push(r);
    }
    console.log(corrTable.toString());
  }
}

// --- Main ---
function main() {
  let filePath = process.argv[2];

  if (!filePath) {
    console.log("[INFO] No input file provided. Generating sample dataset...");
    filePath = "sample_data.csv";
    generateSampleDataset(filePath);
  }

  if (!fs.existsSync(filePath)) {
    console.error(`[ERROR] File not found: ${filePath}`);
    process.exit(1);
  }

  console.log(`[INFO] Loading file: ${filePath}`);
  const data = loadData(filePath);
  const columns = Object.keys(data[0] || {});

  console.log(`[INFO] Shape: ${data.length} rows x ${columns.length} columns\n`);

  // Detect types
  const colTypes = {};
  for (const col of columns) {
    const values = data.map((r) => r[col]);
    colTypes[col] = detectColumnType(values);
  }

  // Profile columns
  const columnProfiles = {};
  for (const col of columns) {
    const values = data.map((r) => r[col]);
    const total = values.length;
    const missingCount = values.filter(isNull).length;
    const missingPct = total > 0 ? round((missingCount / total) * 100, 2) : 0;
    const uniqueCount = new Set(cleanValues(values).map(String)).size;

    const profile = {
      name: col,
      detected_type: colTypes[col],
      total_count: total,
      missing_count: missingCount,
      missing_percentage: missingPct,
      unique_count: uniqueCount,
    };

    if (["integer", "float", "numeric_mixed"].includes(colTypes[col])) {
      profile.numeric_stats = computeNumericStats(values);
    } else if (["categorical", "boolean", "text"].includes(colTypes[col])) {
      profile.categorical_stats = computeCategoricalStats(values);
    }

    profile.distribution = computeDistribution(values, colTypes[col]);
    columnProfiles[col] = profile;
  }

  // Correlation matrix
  const corrMatrix = computeCorrelationMatrix(data, columns, colTypes);

  // Quality score
  const qualityScore = computeDataQualityScore(data, columns, colTypes);

  // Build report
  let totalMissing = 0;
  for (const col of columns) {
    totalMissing += data.map((r) => r[col]).filter(isNull).length;
  }

  const report = {
    file: path.basename(filePath),
    generated_at: new Date().toISOString(),
    dataset_overview: {
      rows: data.length,
      columns: columns.length,
      total_cells: data.length * columns.length,
      total_missing: totalMissing,
      total_missing_pct: round((totalMissing / (data.length * columns.length)) * 100, 2),
    },
    data_quality_score: qualityScore,
    column_profiles: columnProfiles,
    correlation_matrix: corrMatrix,
  };

  // Console output
  printConsoleReport(report);

  // Save JSON report
  const outputDir = path.dirname(path.resolve(filePath));
  const reportPath = path.join(outputDir, "profile_report.json");
  fs.writeFileSync(reportPath, JSON.stringify(report, null, 2), "utf-8");
  console.log(`\n[INFO] Profile report saved: ${reportPath}`);
}

main();
README.md
# Data Profiling Tool - JavaScript Trial 1

## Dependencies
- csv-parse 5.5.6
- simple-statistics 7.8.3
- cli-table3 0.6.5

## Setup
```bash
npm install
```

## Usage
```bash
# Profile a CSV file
node profiler.js data.csv

# Profile a JSON file
node profiler.js data.json

# Generate and profile sample dataset
node profiler.js
```

## Output
- Formatted console report with cli-table3 tables
- `profile_report.json` with full profiling results