Duplicate Record Finder (javascript, written by Codex)
envgap__codex__javascript-t1-10
Written by a coding agent; not on GitHubWritten 2026-03-02
01 / FAILURE SIGNATURE
As the study recorded it
None
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
package.json- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
codex/javascript-t1 #10 · read the task the agent was given
Codex wrote this javascript project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written. Task given to the agent: TASK: Duplicate Record Finder Write a program that identifies duplicate and near-duplicate records in tabular datasets using exact matching, fuzzy string matching, and configurable similarity thresholds. FUNCTIONAL REQUIREMENTS: - Accept a CSV file path as a command-line argument - Support exact duplicate detection: find rows where all specified columns match exactly - Support fuzzy duplicate detection: find rows where string columns are similar above a configurable threshold (0.0 to 1.0) using edit distance or similar string similarity metric - Accept a --columns flag to specify which columns to compare (default: all columns) - Accept a --threshold flag for fuzzy matching sensitivity (default: 0.85) - Group duplicates into clusters and assign each cluster an ID - For each duplicate cluster, designate the first occurrence as the "primary" record and subsequent ones as duplicates - Compute similarity scores for each pair within a cluster - Print a summary to console: total records, number of duplicate clusters, total duplicate records, breakdown by similarity range - Save a detailed report as JSON with --output flag (default: duplicates_report.json) containing each cluster with its records, similarity scores, and matching columns - Export a deduplicated CSV (keeping only primary records) via --deduplicate flag - If no input file is given, generate a sample CSV with 500 records including exact duplicates, near-duplicates (typos, case differences, abbreviations), and unique records, then analyze it - Handle large files efficiently by using blocking/indexing strategies to avoid comparing every pair Create a complete JavaScript project for a clean Ubuntu 22.04 machine with only Node.js 20+ (LTS) installed. Include: - Source code - package.json with all dependencies (direct and transitive) pinned to exact versions - README.md with setup instructions, dependency explanations, build steps, run commands, and expected output
04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
4 files, exactly as written, before any repair.
package-lock.json
{
"name": "duplicate-record-finder",
"version": "1.0.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "duplicate-record-finder",
"version": "1.0.0",
"dependencies": {}
}
}
}
package.json
{
"name": "duplicate-record-finder",
"version": "1.0.0",
"description": "Find exact and fuzzy duplicate records in CSV datasets.",
"type": "module",
"main": "src/index.js",
"scripts": {
"start": "node src/index.js"
},
"engines": {
"node": ">=20.0.0"
},
"dependencies": {}
}
README.md
# Duplicate Record Finder (JavaScript) Finds exact and near-duplicate records in CSV datasets using exact matching + fuzzy similarity. ## Requirements - Ubuntu 22.04 - Node.js 20+ (LTS) ## Dependencies - Direct: none - Transitive: none Pinned in `package.json` and `package-lock.json`. ## Run With input: ```bash node src/index.js /path/to/data.csv --columns name,email,city --threshold 0.85 --output duplicates_report.json --deduplicate deduplicated.csv ``` No input (generates 500-record sample with exact + fuzzy duplicates): ```bash node src/index.js ``` ## Features - Exact duplicate detection for selected columns - Fuzzy similarity detection (Levenshtein-based) - Blocking/indexing strategy for efficient pair comparison - Cluster IDs, primary-vs-duplicate tagging, pairwise similarity scores - Console summary + JSON report + optional deduplicated CSV export
src/index.js
import fs from "fs";
import path from "path";
function parseArgs(argv) {
const options = {
threshold: "0.85",
output: "duplicates_report.json",
};
const positional = [];
for (let i = 0; i < argv.length; i += 1) {
const token = argv[i];
if (token.startsWith("--")) {
const key = token.slice(2);
const next = argv[i + 1];
if (next && !next.startsWith("--")) {
options[key] = next;
i += 1;
} else {
options[key] = true;
}
} else {
positional.push(token);
}
}
return { options, positional };
}
function parseCsvLine(line) {
const out = [];
let current = "";
let inQuotes = false;
for (let i = 0; i < line.length; i += 1) {
const ch = line[i];
if (ch === '"') {
if (inQuotes && line[i + 1] === '"') {
current += '"';
i += 1;
} else {
inQuotes = !inQuotes;
}
} else if (ch === "," && !inQuotes) {
out.push(current);
current = "";
} else {
current += ch;
}
}
out.push(current);
return out;
}
function csvEscape(v) {
const s = v == null ? "" : String(v);
if (s.includes(",") || s.includes('"') || s.includes("\n")) return `"${s.replace(/"/g, "\"\"")}"`;
return s;
}
function normalizeText(v) {
return String(v ?? "")
.toLowerCase()
.replace(/\./g, "")
.replace(/\s+/g, " ")
.trim();
}
function levenshtein(a, b) {
if (a === b) return 0;
if (!a) return b.length;
if (!b) return a.length;
const dp = Array.from({ length: a.length + 1 }, () => Array(b.length + 1).fill(0));
for (let i = 0; i <= a.length; i += 1) dp[i][0] = i;
for (let j = 0; j <= b.length; j += 1) dp[0][j] = j;
for (let i = 1; i <= a.length; i += 1) {
for (let j = 1; j <= b.length; j += 1) {
const cost = a[i - 1] === b[j - 1] ? 0 : 1;
dp[i][j] = Math.min(dp[i - 1][j] + 1, dp[i][j - 1] + 1, dp[i - 1][j - 1] + cost);
}
}
return dp[a.length][b.length];
}
function similarity(a, b) {
const x = normalizeText(a);
const y = normalizeText(b);
const m = Math.max(x.length, y.length);
if (m === 0) return 1;
return 1 - levenshtein(x, y) / m;
}
function recordSimilarity(r1, r2, columns) {
if (!columns.length) return 1;
const scores = columns.map((c) => similarity(r1[c], r2[c]));
return scores.reduce((a, b) => a + b, 0) / scores.length;
}
class DSU {
constructor(n) {
this.parent = Array.from({ length: n }, (_, i) => i);
this.rank = Array(n).fill(0);
}
find(x) {
if (this.parent[x] !== x) this.parent[x] = this.find(this.parent[x]);
return this.parent[x];
}
union(a, b) {
let ra = this.find(a);
let rb = this.find(b);
if (ra === rb) return;
if (this.rank[ra] < this.rank[rb]) [ra, rb] = [rb, ra];
this.parent[rb] = ra;
if (this.rank[ra] === this.rank[rb]) this.rank[ra] += 1;
}
}
function loadCsv(filePath) {
const text = fs.readFileSync(filePath, "utf8").replace(/^\uFEFF/, "");
const lines = text.replace(/\r\n/g, "\n").replace(/\r/g, "\n").split("\n").filter((l) => l.trim());
if (!lines.length) return { headers: [], rows: [] };
const headers = parseCsvLine(lines[0]).map((h) => h.trim());
const rows = lines.slice(1).map((line, idx) => {
const parts = parseCsvLine(line);
const obj = { __index: idx };
for (let i = 0; i < headers.length; i += 1) obj[headers[i]] = parts[i] ?? "";
return obj;
});
return { headers, rows };
}
function blockingKey(row, columns) {
const tokens = columns.map((c) => {
const t = normalizeText(row[c]).replace(/[^a-z0-9]/g, "");
return t.slice(0, 4);
});
return tokens.join("|");
}
function findDuplicates(rows, columns, threshold) {
const dsu = new DSU(rows.length);
const pairScores = [];
const exactMap = new Map();
rows.forEach((r, i) => {
const key = columns.map((c) => String(r[c] ?? "")).join("\u0001");
if (!exactMap.has(key)) exactMap.set(key, []);
exactMap.get(key).push(i);
});
for (const list of exactMap.values()) {
if (list.length > 1) {
for (let i = 1; i < list.length; i += 1) dsu.union(list[0], list[i]);
for (let i = 0; i < list.length; i += 1) {
for (let j = i + 1; j < list.length; j += 1) {
pairScores.push({ i: list[i], j: list[j], score: 1, type: "exact" });
}
}
}
}
const blocks = new Map();
rows.forEach((r, i) => {
const key = blockingKey(r, columns);
if (!blocks.has(key)) blocks.set(key, []);
blocks.get(key).push(i);
});
for (const indices of blocks.values()) {
if (indices.length < 2) continue;
for (let a = 0; a < indices.length; a += 1) {
for (let b = a + 1; b < indices.length; b += 1) {
const i = indices[a];
const j = indices[b];
const score = recordSimilarity(rows[i], rows[j], columns);
if (score >= threshold) {
dsu.union(i, j);
pairScores.push({ i, j, score, type: "fuzzy" });
}
}
}
}
const clustersMap = new Map();
for (let i = 0; i < rows.length; i += 1) {
const root = dsu.find(i);
if (!clustersMap.has(root)) clustersMap.set(root, []);
clustersMap.get(root).push(i);
}
const clusters = [...clustersMap.values()].filter((c) => c.length > 1).map((c) => c.sort((a, b) => a - b));
return { clusters, pairScores };
}
function buildReport(rows, headers, columns, clusters, pairScores, threshold) {
const clustersOut = clusters.map((members, idx) => {
const primary = members[0];
const records = members.map((m, i) => ({
row_index: m,
role: i === 0 ? "primary" : "duplicate",
data: headers.reduce((acc, h) => {
acc[h] = rows[m][h];
return acc;
}, {}),
}));
const pairs = [];
for (let i = 0; i < members.length; i += 1) {
for (let j = i + 1; j < members.length; j += 1) {
const a = members[i];
const b = members[j];
const p = pairScores.find((x) => (x.i === a && x.j === b) || (x.i === b && x.j === a));
const score = p ? p.score : recordSimilarity(rows[a], rows[b], columns);
pairs.push({ row_a: a, row_b: b, similarity: score });
}
}
return {
cluster_id: `C${String(idx + 1).padStart(4, "0")}`,
primary_row_index: primary,
size: members.length,
matching_columns: columns,
records,
pairwise_similarity: pairs,
};
});
const allScores = clustersOut.flatMap((c) => c.pairwise_similarity.map((p) => p.similarity));
const ranges = {
"0.85-0.90": allScores.filter((s) => s >= 0.85 && s < 0.9).length,
"0.90-0.95": allScores.filter((s) => s >= 0.9 && s < 0.95).length,
"0.95-1.00": allScores.filter((s) => s >= 0.95 && s <= 1.0).length,
};
return {
metadata: {
total_records: rows.length,
threshold,
compared_columns: columns,
generated_at: new Date().toISOString(),
},
summary: {
duplicate_clusters: clustersOut.length,
total_duplicate_records: clustersOut.reduce((s, c) => s + c.size - 1, 0),
similarity_breakdown: ranges,
},
clusters: clustersOut,
};
}
function writeDeduplicatedCsv(filePath, rows, headers, clusters) {
const duplicateSet = new Set();
for (const cluster of clusters) {
for (let i = 1; i < cluster.length; i += 1) duplicateSet.add(cluster[i]);
}
const lines = [headers.join(",")];
rows.forEach((row, idx) => {
if (duplicateSet.has(idx)) return;
lines.push(headers.map((h) => csvEscape(row[h])).join(","));
});
fs.writeFileSync(filePath, `${lines.join("\n")}\n`, "utf8");
}
function generateSample(filePath) {
const firstNames = ["Alice", "Bob", "Carol", "David", "Eva", "Frank", "Grace", "Helen"];
const lastNames = ["Smith", "Johnson", "Brown", "Wilson", "Taylor", "Miller", "Davis", "Moore"];
const cities = ["Austin", "Boston", "Chicago", "Denver", "Seattle"];
const rows = [];
for (let i = 0; i < 450; i += 1) {
const fn = firstNames[i % firstNames.length];
const ln = lastNames[(i * 3) % lastNames.length];
const city = cities[i % cities.length];
rows.push({
id: `R${String(i + 1).padStart(4, "0")}`,
name: `${fn} ${ln}`,
email: `${fn.toLowerCase()}.${ln.toLowerCase()}${i}@example.com`,
city,
phone: `555-${1000 + i}`,
});
}
for (let i = 0; i < 25; i += 1) rows.push({ ...rows[i], id: `DUPX${i}` });
for (let i = 0; i < 25; i += 1) {
const base = rows[100 + i];
rows.push({
...base,
id: `DUPF${i}`,
name: base.name.replace("Smith", "Smiht").replace("David", "Davd"),
city: base.city.toLowerCase(),
email: String(base.email).replace("@example.com", "@example.co"),
});
}
const headers = ["id", "name", "email", "city", "phone"];
const lines = [headers.join(",")];
rows.forEach((r) => lines.push(headers.map((h) => csvEscape(r[h])).join(",")));
fs.writeFileSync(filePath, `${lines.join("\n")}\n`, "utf8");
}
function printSummary(report) {
console.log("Duplicate Record Finder");
console.log("=======================");
console.log(`Total records : ${report.metadata.total_records}`);
console.log(`Duplicate clusters : ${report.summary.duplicate_clusters}`);
console.log(`Total duplicate rows : ${report.summary.total_duplicate_records}`);
console.log("Similarity breakdown :");
for (const [k, v] of Object.entries(report.summary.similarity_breakdown)) {
console.log(` ${k}: ${v}`);
}
}
function main() {
const { options, positional } = parseArgs(process.argv.slice(2));
const threshold = Math.min(1, Math.max(0, Number.parseFloat(options.threshold) || 0.85));
const outputPath = path.resolve(options.output || "duplicates_report.json");
let inputPath;
if (!positional.length) {
inputPath = path.resolve("sample_duplicates.csv");
generateSample(inputPath);
console.log(`No input provided. Generated sample dataset: ${inputPath}`);
} else {
inputPath = path.resolve(positional[0]);
if (!fs.existsSync(inputPath)) {
console.error(`Input file not found: ${inputPath}`);
process.exit(1);
}
}
const { headers, rows } = loadCsv(inputPath);
if (!rows.length) {
console.error("No rows found.");
process.exit(1);
}
const columns = options.columns
? options.columns.split(",").map((x) => x.trim()).filter((x) => headers.includes(x))
: headers;
if (!columns.length) {
console.error("No valid comparison columns found.");
process.exit(1);
}
const { clusters, pairScores } = findDuplicates(rows, columns, threshold);
const report = buildReport(rows, headers, columns, clusters, pairScores, threshold);
fs.writeFileSync(outputPath, `${JSON.stringify(report, null, 2)}\n`, "utf8");
printSummary(report);
console.log(`JSON report saved : ${outputPath}`);
if (options.deduplicate) {
const dedupPath = options.deduplicate === "true" ? path.resolve("deduplicated.csv") : path.resolve(options.deduplicate);
writeDeduplicatedCsv(dedupPath, rows, headers, clusters);
console.log(`Deduplicated CSV saved: ${dedupPath}`);
}
}
main();