← All tasks
javascriptclaude-code/javascript-t1 #32Not a task: already works

TF-IDF Search Engine (javascript, written by Claude Code)

envgap__claude-code__javascript-t1-32

Written by a coding agent; not on GitHubWritten 2026-02-27

01 / FAILURE SIGNATURE

As the study recorded it

No identifying execution failure has been captured.
Not a benchmark task.
  • The project already builds and runs before the fix, so there is nothing to repair.

02 / ENVIRONMENT RECIPE

Base commit
Not freshly verified
Manifest
package.json
Reproduce
Awaiting issue-specific recipe
Run under trace
Awaiting a meaningful runtime command

03 / TASK AND FAILURE

claude-code/javascript-t1 #32 · read the task the agent was given
Claude Code wrote this javascript project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written.

Task given to the agent:

TASK: TF-IDF Search Engine

Write a program that builds a TF-IDF (Term Frequency-Inverse Document Frequency) index over a collection of text documents and supports ranked keyword search queries returning the most relevant documents.

FUNCTIONAL REQUIREMENTS:
- Accept a directory of text files as a command-line argument to build the index
- Tokenize documents: split on whitespace and punctuation, convert to lowercase, remove stop words (built-in list of common English stop words like "the", "is", "and", etc.)
- Support optional stemming/lemmatization via --stem flag to group word variants (e.g., "running", "runs", "ran" all map to "run")
- Compute TF-IDF scores for each term in each document using standard formulas: TF = term count / total terms in document, IDF = log(total documents / documents containing term)
- Accept search queries via --query flag and return the top N most relevant documents ranked by cosine similarity between query vector and document vectors (--top flag, default 10)
- Support multi-word queries: compute a query TF-IDF vector and rank documents by similarity
- Support boolean operators in queries via --boolean flag: AND (both terms required), OR (either term), NOT (exclude term)
- Display search results showing: rank, document name, relevance score, and a snippet of the matching text with query terms highlighted
- Save the built index to a file via --save-index flag for reuse without reprocessing
- Load a previously saved index via --load-index flag
- Print index statistics: total documents, total unique terms, average document length, most common terms (top 20)
- Save search results as JSON with --output flag
- If no directory is given, generate a sample corpus of 20 short documents on varied topics (science, sports, technology, cooking, travel), build the index, and demonstrate several search queries with ranked results
- Handle errors: empty documents, binary files in the directory, extremely large documents, and empty queries

Create a complete JavaScript project for a clean Ubuntu 22.04 machine with only Node.js 20+ (LTS) installed. Include:
- Source code
- package.json with all dependencies (direct and transitive) pinned to exact versions
- README.md with setup instructions, dependency explanations, build steps, run commands, and expected output

04 / LABELS

Labels from the report text only; not yet run

No supported category has been assigned.

Label rules and the text that matched
[]

05 / FILES

The project as the agent wrote it

2 files, exactly as written, before any repair.

package.json
{
  "name": "tfidf-search-engine",
  "version": "1.0.0",
  "description": "TF-IDF search engine that builds an index over documents and supports ranked keyword search with cosine similarity",
  "main": "search.js",
  "scripts": {
    "start": "node search.js",
    "demo": "node search.js --demo",
    "interactive": "node search.js --demo --interactive"
  },
  "keywords": ["tfidf", "search", "nlp", "cosine-similarity"],
  "license": "MIT",
  "dependencies": {
    "natural": "6.10.4",
    "chalk": "4.1.2"
  }
}
search.js
/**
 * TF-IDF Search Engine
 *
 * Builds a TF-IDF index over a collection of documents and supports
 * ranked keyword search with cosine similarity.
 *
 * Dependencies: natural (6.10.4), chalk (4.1.2)
 */

const fs = require("fs");
const path = require("path");
const natural = require("natural");
const chalk = require("chalk");

/**
 * TF-IDF Search Engine using the natural NLP library.
 */
class TFIDFSearchEngine {
  constructor(options = {}) {
    this.useStemming = options.useStemming !== false;
    this.tokenizer = new natural.WordTokenizer();
    this.tfidf = new natural.TfIdf();
    this.documents = [];
    this.docNames = [];
    this.stopWords = new Set([
      "a", "an", "the", "is", "are", "was", "were", "be", "been", "being",
      "have", "has", "had", "do", "does", "did", "will", "would", "could",
      "should", "may", "might", "shall", "can", "need", "dare", "ought",
      "used", "to", "of", "in", "for", "on", "with", "at", "by", "from",
      "as", "into", "through", "during", "before", "after", "above", "below",
      "between", "out", "off", "over", "under", "again", "further", "then",
      "once", "here", "there", "when", "where", "why", "how", "all", "both",
      "each", "few", "more", "most", "other", "some", "such", "no", "nor",
      "not", "only", "own", "same", "so", "than", "too", "very", "just",
      "because", "but", "and", "or", "if", "while", "about", "up", "it",
      "its", "this", "that", "these", "those", "i", "me", "my", "we", "our",
      "you", "your", "he", "him", "his", "she", "her", "they", "them", "their",
      "what", "which", "who", "whom",
    ]);
  }

  /**
   * Preprocess text by tokenizing, removing stopwords, and optionally stemming.
   * @param {string} text - Input text.
   * @returns {string} Processed text.
   */
  preprocessText(text) {
    let tokens = this.tokenizer.tokenize(text.toLowerCase());
    tokens = tokens.filter(
      (t) => t.length > 1 && !this.stopWords.has(t) && /^[a-z0-9]+$/.test(t)
    );
    if (this.useStemming) {
      tokens = tokens.map((t) => natural.PorterStemmer.stem(t));
    }
    return tokens.join(" ");
  }

  /**
   * Add a document to the collection.
   * @param {string} name - Document identifier.
   * @param {string} content - Document text content.
   */
  addDocument(name, content) {
    const processed = this.preprocessText(content);
    this.tfidf.addDocument(processed);
    this.documents.push(content);
    this.docNames.push(name);
  }

  /**
   * Load all .txt files from a directory.
   * @param {string} dirPath - Path to the directory.
   */
  loadFromDirectory(dirPath) {
    if (!fs.existsSync(dirPath) || !fs.statSync(dirPath).isDirectory()) {
      throw new Error(`Directory not found: ${dirPath}`);
    }

    const files = fs
      .readdirSync(dirPath)
      .filter((f) => f.endsWith(".txt"))
      .sort();

    for (const file of files) {
      const content = fs.readFileSync(path.join(dirPath, file), "utf-8");
      this.addDocument(file, content);
    }

    console.log(
      chalk.green(`Loaded ${files.length} document(s) from '${dirPath}'.`)
    );
  }

  /**
   * Compute cosine similarity between two term frequency vectors.
   * @param {Object} vecA - First vector as {term: weight}.
   * @param {Object} vecB - Second vector as {term: weight}.
   * @returns {number} Cosine similarity score.
   */
  cosineSimilarity(vecA, vecB) {
    let dotProduct = 0;
    let normA = 0;
    let normB = 0;

    const allTerms = new Set([...Object.keys(vecA), ...Object.keys(vecB)]);
    for (const term of allTerms) {
      const a = vecA[term] || 0;
      const b = vecB[term] || 0;
      dotProduct += a * b;
      normA += a * a;
      normB += b * b;
    }

    if (normA === 0 || normB === 0) return 0;
    return dotProduct / (Math.sqrt(normA) * Math.sqrt(normB));
  }

  /**
   * Search the index with a keyword query.
   * @param {string} query - The search query.
   * @param {number} topK - Number of top results to return.
   * @returns {Array<{name: string, score: number}>} Ranked results.
   */
  search(query, topK = 5) {
    if (this.documents.length === 0) {
      throw new Error("No documents indexed. Add documents first.");
    }

    const results = [];

    this.tfidf.tfidfs(this.preprocessText(query), (docIndex, measure) => {
      results.push({
        name: this.docNames[docIndex],
        score: measure,
      });
    });

    // Sort by descending score and take top-k
    results.sort((a, b) => b.score - a.score);
    return results.filter((r) => r.score > 0).slice(0, topK);
  }

  /**
   * Get the number of indexed documents.
   * @returns {number}
   */
  getDocumentCount() {
    return this.documents.length;
  }
}

/**
 * Create built-in sample documents for demonstration.
 * @returns {Array<{name: string, content: string}>}
 */
function createSampleDocuments() {
  return [
    {
      name: "python_intro.txt",
      content:
        "Python is a high-level programming language known for its simplicity " +
        "and readability. It supports multiple programming paradigms including " +
        "object-oriented, procedural, and functional programming. Python has a " +
        "large standard library and an active community.",
    },
    {
      name: "machine_learning.txt",
      content:
        "Machine learning is a subset of artificial intelligence that enables " +
        "systems to learn from data. Supervised learning uses labeled data to " +
        "train models, while unsupervised learning finds patterns in unlabeled " +
        "data. Deep learning uses neural networks with many layers.",
    },
    {
      name: "web_development.txt",
      content:
        "Web development encompasses building websites and web applications. " +
        "Frontend development focuses on the user interface using HTML, CSS, " +
        "and JavaScript. Backend development handles server-side logic, " +
        "databases, and APIs.",
    },
    {
      name: "data_science.txt",
      content:
        "Data science combines statistics, mathematics, and computer science " +
        "to extract insights from data. Common tools include Python, R, and " +
        "SQL. Data visualization helps communicate findings effectively.",
    },
    {
      name: "algorithms.txt",
      content:
        "Algorithms are step-by-step procedures for solving problems. Sorting " +
        "algorithms like quicksort and mergesort organize data efficiently. " +
        "Search algorithms such as binary search find elements in collections. " +
        "Graph algorithms traverse networks and find shortest paths.",
    },
  ];
}

/**
 * Print search results with colored formatting.
 */
function printResults(query, results) {
  if (results.length > 0) {
    console.log(
      chalk.cyan(`\nTop ${results.length} result(s) for '${query}':`)
    );
    results.forEach((r, i) => {
      console.log(
        `  ${chalk.yellow(i + 1 + ".")} ${chalk.white(r.name)} ${chalk.gray(
          "(score: " + r.score.toFixed(4) + ")"
        )}`
      );
    });
  } else {
    console.log(chalk.red(`No results found for '${query}'.`));
  }
}

/**
 * Run interactive search mode.
 */
async function interactiveMode(engine, topK) {
  const readline = require("readline");
  const rl = readline.createInterface({
    input: process.stdin,
    output: process.stdout,
  });

  console.log(
    chalk.cyan("Interactive search mode. Type 'quit' or 'exit' to stop.\n")
  );

  const askQuery = () => {
    rl.question(chalk.green("Query> "), (line) => {
      const trimmed = line.trim();
      if (
        trimmed.toLowerCase() === "quit" ||
        trimmed.toLowerCase() === "exit" ||
        trimmed.toLowerCase() === "q"
      ) {
        console.log("Exiting.");
        rl.close();
        return;
      }
      if (trimmed.length === 0) {
        askQuery();
        return;
      }

      const results = engine.search(trimmed, topK);
      printResults(trimmed, results);
      console.log();
      askQuery();
    });
  };

  askQuery();
}

/**
 * Parse command-line arguments.
 */
function parseArgs(argv) {
  const args = {
    dir: null,
    query: null,
    topK: 5,
    interactive: false,
    demo: false,
    noStemming: false,
  };

  for (let i = 2; i < argv.length; i++) {
    switch (argv[i]) {
      case "--dir":
        args.dir = argv[++i];
        break;
      case "--query":
      case "-q":
        args.query = argv[++i];
        break;
      case "--top-k":
      case "-k":
        args.topK = parseInt(argv[++i], 10);
        break;
      case "--interactive":
      case "-i":
        args.interactive = true;
        break;
      case "--demo":
        args.demo = true;
        break;
      case "--no-stemming":
        args.noStemming = true;
        break;
      default:
        console.error(chalk.red(`Unknown argument: ${argv[i]}`));
        process.exit(1);
    }
  }

  return args;
}

// --- Main ---

function main() {
  const args = parseArgs(process.argv);
  const engine = new TFIDFSearchEngine({ useStemming: !args.noStemming });

  // Load documents
  if (args.dir) {
    engine.loadFromDirectory(args.dir);
  } else if (args.demo || (!args.query && !args.interactive)) {
    console.log(chalk.cyan("Running with sample documents...\n"));
    const samples = createSampleDocuments();
    for (const doc of samples) {
      engine.addDocument(doc.name, doc.content);
    }
  } else {
    console.error(
      chalk.red("No document source specified. Use --dir or --demo.")
    );
    process.exit(1);
  }

  console.log(
    chalk.green(`Index built: ${engine.getDocumentCount()} documents.\n`)
  );

  // Execute search
  if (args.interactive) {
    interactiveMode(engine, args.topK);
  } else if (args.query) {
    const results = engine.search(args.query, args.topK);
    printResults(args.query, results);
  } else {
    // Default demo queries
    const sampleQueries = [
      "machine learning neural networks",
      "python programming language",
      "web frontend JavaScript",
      "sorting search algorithms",
    ];

    for (const q of sampleQueries) {
      const results = engine.search(q, 3);
      printResults(q, results);
      console.log();
    }
  }
}

main();