← All tasks
javascriptclaude-code/javascript-t1 #39Not a task: already works

File Deduplicator (javascript, written by Claude Code)

envgap__claude-code__javascript-t1-39

Written by a coding agent; not on GitHubWritten 2026-02-27

01 / FAILURE SIGNATURE

As the study recorded it

No identifying execution failure has been captured.
Not a benchmark task.
  • The project already builds and runs before the fix, so there is nothing to repair.

02 / ENVIRONMENT RECIPE

Base commit
Not freshly verified
Manifest
package.json
Reproduce
Awaiting issue-specific recipe
Run under trace
Awaiting a meaningful runtime command

03 / TASK AND FAILURE

claude-code/javascript-t1 #39 · read the task the agent was given
Claude Code wrote this javascript project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written.

Task given to the agent:

TASK: File Deduplicator

Write a program that finds and manages duplicate files across directories using content-based hashing, supporting multiple deduplication strategies and detailed reporting.

FUNCTIONAL REQUIREMENTS:
- Accept one or more directory paths as command-line arguments
- Find duplicate files by comparing SHA-256 content hashes, using a two-phase approach: first compare file sizes to narrow candidates, then hash only size-matched files
- Support configurable minimum file size via --min-size flag (default: 1 byte) to skip tiny files
- Support file type filtering via --include and --exclude flags with glob patterns
- Group duplicates into sets showing all copies with their full paths, sizes, and modification dates
- Support multiple deduplication actions via --action flag: report (default, just list duplicates), delete (remove duplicates keeping the oldest/newest based on --keep flag), hardlink (replace duplicates with hard links to save space), symlink (replace with symbolic links)
- Support a --dry-run flag to preview what would be done without actually modifying files
- Scan directories recursively by default, with --no-recursive flag to disable
- Display a progress bar during scanning showing files processed and duplicates found so far
- Print summary to console: total files scanned, total unique files, duplicate sets found, total wasted space, space that would be recovered
- Save the full deduplication report as JSON with --output flag (default: dedup_report.json)
- If no directories are given, create a sample directory with intentional duplicates (exact copies, files with same content but different names, and unique files), run deduplication analysis, and display the results
- Handle errors: permission denied, broken symlinks, files modified during scan, and cross-filesystem hard links

Create a complete JavaScript project for a clean Ubuntu 22.04 machine with only Node.js 20+ (LTS) installed. Include:
- Source code
- package.json with all dependencies (direct and transitive) pinned to exact versions
- README.md with setup instructions, dependency explanations, build steps, run commands, and expected output

04 / LABELS

Labels from the report text only; not yet run

No supported category has been assigned.

Label rules and the text that matched
[]

05 / FILES

The project as the agent wrote it

2 files, exactly as written, before any repair.

dedup.js
#!/usr/bin/env node
/**
 * File Deduplicator - Finds duplicate files via content hashing.
 * Supports hardlink, symlink, and delete deduplication strategies.
 *
 * Uses Node.js built-in crypto for hashing, glob for file discovery,
 * and chalk for colored console output.
 */

const fs = require('fs');
const path = require('path');
const crypto = require('crypto');
const { globSync } = require('glob');
const chalk = require('chalk');

/**
 * Compute SHA-256 hash of a file's contents.
 */
function computeHash(filePath) {
    return new Promise((resolve, reject) => {
        const hash = crypto.createHash('sha256');
        const stream = fs.createReadStream(filePath);
        stream.on('data', (chunk) => hash.update(chunk));
        stream.on('end', () => resolve(hash.digest('hex')));
        stream.on('error', (err) => {
            console.error(chalk.yellow(`Warning: Cannot read ${filePath}: ${err.message}`));
            resolve(null);
        });
    });
}

/**
 * Group files by their size as a preliminary filter.
 */
function groupBySize(directory, recursive, minSize) {
    const pattern = recursive ? '**/*' : '*';
    const files = globSync(pattern, {
        cwd: directory,
        absolute: true,
        nodir: true,
        follow: false,
    });

    const sizeMap = new Map();
    for (const filePath of files) {
        try {
            const stats = fs.lstatSync(filePath);
            if (stats.isFile() && !stats.isSymbolicLink() && stats.size >= minSize) {
                const size = stats.size;
                if (!sizeMap.has(size)) {
                    sizeMap.set(size, []);
                }
                sizeMap.get(size).push(filePath);
            }
        } catch (err) {
            // Skip inaccessible files
        }
    }

    // Only keep sizes with more than one file
    for (const [size, paths] of sizeMap) {
        if (paths.length < 2) {
            sizeMap.delete(size);
        }
    }
    return sizeMap;
}

/**
 * Find duplicate files by first grouping by size, then hashing.
 */
async function findDuplicates(directory, recursive, minSize) {
    console.log(chalk.blue('Phase 1: Grouping files by size...'));
    const sizeGroups = groupBySize(directory, recursive, minSize);

    let candidateCount = 0;
    for (const paths of sizeGroups.values()) {
        candidateCount += paths.length;
    }
    console.log(`  Found ${candidateCount} candidate files in ${sizeGroups.size} size groups.`);

    console.log(chalk.blue('Phase 2: Hashing file contents...'));
    const hashMap = new Map();
    let processed = 0;

    for (const paths of sizeGroups.values()) {
        for (const filePath of paths) {
            const hash = await computeHash(filePath);
            if (hash !== null) {
                if (!hashMap.has(hash)) {
                    hashMap.set(hash, []);
                }
                hashMap.get(hash).push(filePath);
            }
            processed++;
            if (processed % 100 === 0) {
                process.stdout.write(`  Hashed ${processed} / ${candidateCount} files...\r`);
            }
        }
    }
    if (candidateCount > 0) {
        console.log(`  Hashed ${processed} / ${candidateCount} files.`);
    }

    // Only keep hashes with more than one file
    for (const [hash, paths] of hashMap) {
        if (paths.length < 2) {
            hashMap.delete(hash);
        }
    }
    return hashMap;
}

/**
 * Replace duplicate files with hardlinks to the first occurrence.
 */
function deduplicateHardlink(duplicates, dryRun) {
    let saved = 0;
    for (const [hash, paths] of duplicates) {
        const original = paths[0];
        for (let i = 1; i < paths.length; i++) {
            const duplicate = paths[i];
            const size = fs.statSync(duplicate).size;
            if (dryRun) {
                console.log(chalk.cyan(`  [DRY RUN] Would hardlink: ${duplicate} -> ${original}`));
            } else {
                try {
                    fs.unlinkSync(duplicate);
                    fs.linkSync(original, duplicate);
                    console.log(chalk.green(`  Hardlinked: ${duplicate} -> ${original}`));
                } catch (err) {
                    console.error(chalk.red(`  Error hardlinking ${duplicate}: ${err.message}`));
                    continue;
                }
            }
            saved += size;
        }
    }
    return saved;
}

/**
 * Replace duplicate files with symlinks to the first occurrence.
 */
function deduplicateSymlink(duplicates, dryRun) {
    let saved = 0;
    for (const [hash, paths] of duplicates) {
        const original = path.resolve(paths[0]);
        for (let i = 1; i < paths.length; i++) {
            const duplicate = paths[i];
            const size = fs.statSync(duplicate).size;
            if (dryRun) {
                console.log(chalk.cyan(`  [DRY RUN] Would symlink: ${duplicate} -> ${original}`));
            } else {
                try {
                    fs.unlinkSync(duplicate);
                    fs.symlinkSync(original, duplicate);
                    console.log(chalk.green(`  Symlinked: ${duplicate} -> ${original}`));
                } catch (err) {
                    console.error(chalk.red(`  Error symlinking ${duplicate}: ${err.message}`));
                    continue;
                }
            }
            saved += size;
        }
    }
    return saved;
}

/**
 * Delete duplicate files, keeping only the first occurrence.
 */
function deduplicateDelete(duplicates, dryRun) {
    let saved = 0;
    for (const [hash, paths] of duplicates) {
        for (let i = 1; i < paths.length; i++) {
            const duplicate = paths[i];
            const size = fs.statSync(duplicate).size;
            if (dryRun) {
                console.log(chalk.cyan(`  [DRY RUN] Would delete: ${duplicate}`));
            } else {
                try {
                    fs.unlinkSync(duplicate);
                    console.log(chalk.green(`  Deleted: ${duplicate}`));
                } catch (err) {
                    console.error(chalk.red(`  Error deleting ${duplicate}: ${err.message}`));
                    continue;
                }
            }
            saved += size;
        }
    }
    return saved;
}

/**
 * Format byte count into human-readable string.
 */
function formatSize(bytes) {
    const units = ['B', 'KB', 'MB', 'GB', 'TB'];
    let size = bytes;
    for (const unit of units) {
        if (size < 1024.0) {
            return `${size.toFixed(2)} ${unit}`;
        }
        size /= 1024.0;
    }
    return `${size.toFixed(2)} PB`;
}

/**
 * Print a report of found duplicates.
 */
function printReport(duplicates) {
    if (duplicates.size === 0) {
        console.log(chalk.yellow('\nNo duplicate files found.'));
        return;
    }

    let totalGroups = duplicates.size;
    let totalFiles = 0;
    let totalWasted = 0;

    for (const [hash, paths] of duplicates) {
        totalFiles += paths.length;
        try {
            const size = fs.statSync(paths[0]).size;
            totalWasted += size * (paths.length - 1);
        } catch (err) {
            // skip
        }
    }

    console.log('\n' + chalk.bold('='.repeat(60)));
    console.log(chalk.bold('Duplicate Report'));
    console.log(chalk.bold('='.repeat(60)));
    console.log(`  Duplicate groups:  ${totalGroups}`);
    console.log(`  Total files:       ${totalFiles}`);
    console.log(`  Wasted space:      ${formatSize(totalWasted)}`);
    console.log(chalk.bold('='.repeat(60)));

    let groupNum = 1;
    for (const [hash, paths] of duplicates) {
        let size = 0;
        try { size = fs.statSync(paths[0]).size; } catch (err) { /* ignore */ }
        console.log(chalk.white(
            `\nGroup ${groupNum} (hash: ${hash.substring(0, 16)}..., size: ${formatSize(size)}):`
        ));
        for (const p of paths) {
            console.log(`    ${p}`);
        }
        groupNum++;
    }
}

/**
 * Parse command-line arguments.
 */
function parseArgs() {
    const args = process.argv.slice(2);
    const opts = {
        directory: null,
        strategy: 'report',
        recursive: true,
        dryRun: false,
        minSize: 1,
    };

    for (let i = 0; i < args.length; i++) {
        switch (args[i]) {
            case '-s':
            case '--strategy':
                opts.strategy = args[++i];
                break;
            case '--no-recursive':
                opts.recursive = false;
                break;
            case '--dry-run':
                opts.dryRun = true;
                break;
            case '--min-size':
                opts.minSize = parseInt(args[++i], 10);
                break;
            case '-h':
            case '--help':
                console.log('Usage: node dedup.js <directory> [options]');
                console.log('Options:');
                console.log('  -s, --strategy <strategy>  report|hardlink|symlink|delete (default: report)');
                console.log('  --no-recursive             Do not scan subdirectories');
                console.log('  --dry-run                  Preview changes without applying');
                console.log('  --min-size <bytes>         Minimum file size (default: 1)');
                process.exit(0);
                break;
            default:
                if (!args[i].startsWith('-')) {
                    opts.directory = args[i];
                }
                break;
        }
    }
    return opts;
}

async function main() {
    const opts = parseArgs();

    if (!opts.directory) {
        console.error(chalk.red('Error: No directory specified.'));
        process.exit(1);
    }

    if (!fs.existsSync(opts.directory) || !fs.statSync(opts.directory).isDirectory()) {
        console.error(chalk.red(`Error: '${opts.directory}' is not a valid directory.`));
        process.exit(1);
    }

    const validStrategies = ['report', 'hardlink', 'symlink', 'delete'];
    if (!validStrategies.includes(opts.strategy)) {
        console.error(chalk.red(`Error: Invalid strategy '${opts.strategy}'.`));
        process.exit(1);
    }

    console.log(`Scanning: ${path.resolve(opts.directory)}`);
    const duplicates = await findDuplicates(opts.directory, opts.recursive, opts.minSize);
    printReport(duplicates);

    if (duplicates.size === 0) {
        process.exit(0);
    }

    const strategies = {
        hardlink: deduplicateHardlink,
        symlink: deduplicateSymlink,
        delete: deduplicateDelete,
    };

    if (strategies[opts.strategy]) {
        const label = opts.dryRun ? ` (dry run)` : '';
        console.log(chalk.blue(`\nApplying strategy: ${opts.strategy}${label}`));
        const saved = strategies[opts.strategy](duplicates, opts.dryRun);
        const prefix = opts.dryRun ? 'that would be ' : '';
        console.log(`\nSpace ${prefix}recovered: ${formatSize(saved)}`);
    }
}

main().catch((err) => {
    console.error(chalk.red(`Error: ${err.message}`));
    process.exit(1);
});
package.json
{
  "name": "file-deduplicator",
  "version": "1.0.0",
  "description": "Finds duplicate files via content hashing with hardlink/symlink/delete deduplication strategies.",
  "main": "dedup.js",
  "scripts": {
    "start": "node dedup.js"
  },
  "dependencies": {
    "glob": "10.3.10",
    "chalk": "4.1.2"
  },
  "engines": {
    "node": ">=16.0.0"
  },
  "license": "MIT"
}