TF-IDF Search Engine (javascript, written by Claude Code)
envgap__claude-code__javascript-t1-32
Written by a coding agent; not on GitHubWritten 2026-02-27
01 / FAILURE SIGNATURE
As the study recorded it
No identifying execution failure has been captured.
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
package.json- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
claude-code/javascript-t1 #32 · read the task the agent was given
Claude Code wrote this javascript project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written. Task given to the agent: TASK: TF-IDF Search Engine Write a program that builds a TF-IDF (Term Frequency-Inverse Document Frequency) index over a collection of text documents and supports ranked keyword search queries returning the most relevant documents. FUNCTIONAL REQUIREMENTS: - Accept a directory of text files as a command-line argument to build the index - Tokenize documents: split on whitespace and punctuation, convert to lowercase, remove stop words (built-in list of common English stop words like "the", "is", "and", etc.) - Support optional stemming/lemmatization via --stem flag to group word variants (e.g., "running", "runs", "ran" all map to "run") - Compute TF-IDF scores for each term in each document using standard formulas: TF = term count / total terms in document, IDF = log(total documents / documents containing term) - Accept search queries via --query flag and return the top N most relevant documents ranked by cosine similarity between query vector and document vectors (--top flag, default 10) - Support multi-word queries: compute a query TF-IDF vector and rank documents by similarity - Support boolean operators in queries via --boolean flag: AND (both terms required), OR (either term), NOT (exclude term) - Display search results showing: rank, document name, relevance score, and a snippet of the matching text with query terms highlighted - Save the built index to a file via --save-index flag for reuse without reprocessing - Load a previously saved index via --load-index flag - Print index statistics: total documents, total unique terms, average document length, most common terms (top 20) - Save search results as JSON with --output flag - If no directory is given, generate a sample corpus of 20 short documents on varied topics (science, sports, technology, cooking, travel), build the index, and demonstrate several search queries with ranked results - Handle errors: empty documents, binary files in the directory, extremely large documents, and empty queries Create a complete JavaScript project for a clean Ubuntu 22.04 machine with only Node.js 20+ (LTS) installed. Include: - Source code - package.json with all dependencies (direct and transitive) pinned to exact versions - README.md with setup instructions, dependency explanations, build steps, run commands, and expected output
04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
2 files, exactly as written, before any repair.
package.json
{
"name": "tfidf-search-engine",
"version": "1.0.0",
"description": "TF-IDF search engine that builds an index over documents and supports ranked keyword search with cosine similarity",
"main": "search.js",
"scripts": {
"start": "node search.js",
"demo": "node search.js --demo",
"interactive": "node search.js --demo --interactive"
},
"keywords": ["tfidf", "search", "nlp", "cosine-similarity"],
"license": "MIT",
"dependencies": {
"natural": "6.10.4",
"chalk": "4.1.2"
}
}
search.js
/**
* TF-IDF Search Engine
*
* Builds a TF-IDF index over a collection of documents and supports
* ranked keyword search with cosine similarity.
*
* Dependencies: natural (6.10.4), chalk (4.1.2)
*/
const fs = require("fs");
const path = require("path");
const natural = require("natural");
const chalk = require("chalk");
/**
* TF-IDF Search Engine using the natural NLP library.
*/
class TFIDFSearchEngine {
constructor(options = {}) {
this.useStemming = options.useStemming !== false;
this.tokenizer = new natural.WordTokenizer();
this.tfidf = new natural.TfIdf();
this.documents = [];
this.docNames = [];
this.stopWords = new Set([
"a", "an", "the", "is", "are", "was", "were", "be", "been", "being",
"have", "has", "had", "do", "does", "did", "will", "would", "could",
"should", "may", "might", "shall", "can", "need", "dare", "ought",
"used", "to", "of", "in", "for", "on", "with", "at", "by", "from",
"as", "into", "through", "during", "before", "after", "above", "below",
"between", "out", "off", "over", "under", "again", "further", "then",
"once", "here", "there", "when", "where", "why", "how", "all", "both",
"each", "few", "more", "most", "other", "some", "such", "no", "nor",
"not", "only", "own", "same", "so", "than", "too", "very", "just",
"because", "but", "and", "or", "if", "while", "about", "up", "it",
"its", "this", "that", "these", "those", "i", "me", "my", "we", "our",
"you", "your", "he", "him", "his", "she", "her", "they", "them", "their",
"what", "which", "who", "whom",
]);
}
/**
* Preprocess text by tokenizing, removing stopwords, and optionally stemming.
* @param {string} text - Input text.
* @returns {string} Processed text.
*/
preprocessText(text) {
let tokens = this.tokenizer.tokenize(text.toLowerCase());
tokens = tokens.filter(
(t) => t.length > 1 && !this.stopWords.has(t) && /^[a-z0-9]+$/.test(t)
);
if (this.useStemming) {
tokens = tokens.map((t) => natural.PorterStemmer.stem(t));
}
return tokens.join(" ");
}
/**
* Add a document to the collection.
* @param {string} name - Document identifier.
* @param {string} content - Document text content.
*/
addDocument(name, content) {
const processed = this.preprocessText(content);
this.tfidf.addDocument(processed);
this.documents.push(content);
this.docNames.push(name);
}
/**
* Load all .txt files from a directory.
* @param {string} dirPath - Path to the directory.
*/
loadFromDirectory(dirPath) {
if (!fs.existsSync(dirPath) || !fs.statSync(dirPath).isDirectory()) {
throw new Error(`Directory not found: ${dirPath}`);
}
const files = fs
.readdirSync(dirPath)
.filter((f) => f.endsWith(".txt"))
.sort();
for (const file of files) {
const content = fs.readFileSync(path.join(dirPath, file), "utf-8");
this.addDocument(file, content);
}
console.log(
chalk.green(`Loaded ${files.length} document(s) from '${dirPath}'.`)
);
}
/**
* Compute cosine similarity between two term frequency vectors.
* @param {Object} vecA - First vector as {term: weight}.
* @param {Object} vecB - Second vector as {term: weight}.
* @returns {number} Cosine similarity score.
*/
cosineSimilarity(vecA, vecB) {
let dotProduct = 0;
let normA = 0;
let normB = 0;
const allTerms = new Set([...Object.keys(vecA), ...Object.keys(vecB)]);
for (const term of allTerms) {
const a = vecA[term] || 0;
const b = vecB[term] || 0;
dotProduct += a * b;
normA += a * a;
normB += b * b;
}
if (normA === 0 || normB === 0) return 0;
return dotProduct / (Math.sqrt(normA) * Math.sqrt(normB));
}
/**
* Search the index with a keyword query.
* @param {string} query - The search query.
* @param {number} topK - Number of top results to return.
* @returns {Array<{name: string, score: number}>} Ranked results.
*/
search(query, topK = 5) {
if (this.documents.length === 0) {
throw new Error("No documents indexed. Add documents first.");
}
const results = [];
this.tfidf.tfidfs(this.preprocessText(query), (docIndex, measure) => {
results.push({
name: this.docNames[docIndex],
score: measure,
});
});
// Sort by descending score and take top-k
results.sort((a, b) => b.score - a.score);
return results.filter((r) => r.score > 0).slice(0, topK);
}
/**
* Get the number of indexed documents.
* @returns {number}
*/
getDocumentCount() {
return this.documents.length;
}
}
/**
* Create built-in sample documents for demonstration.
* @returns {Array<{name: string, content: string}>}
*/
function createSampleDocuments() {
return [
{
name: "python_intro.txt",
content:
"Python is a high-level programming language known for its simplicity " +
"and readability. It supports multiple programming paradigms including " +
"object-oriented, procedural, and functional programming. Python has a " +
"large standard library and an active community.",
},
{
name: "machine_learning.txt",
content:
"Machine learning is a subset of artificial intelligence that enables " +
"systems to learn from data. Supervised learning uses labeled data to " +
"train models, while unsupervised learning finds patterns in unlabeled " +
"data. Deep learning uses neural networks with many layers.",
},
{
name: "web_development.txt",
content:
"Web development encompasses building websites and web applications. " +
"Frontend development focuses on the user interface using HTML, CSS, " +
"and JavaScript. Backend development handles server-side logic, " +
"databases, and APIs.",
},
{
name: "data_science.txt",
content:
"Data science combines statistics, mathematics, and computer science " +
"to extract insights from data. Common tools include Python, R, and " +
"SQL. Data visualization helps communicate findings effectively.",
},
{
name: "algorithms.txt",
content:
"Algorithms are step-by-step procedures for solving problems. Sorting " +
"algorithms like quicksort and mergesort organize data efficiently. " +
"Search algorithms such as binary search find elements in collections. " +
"Graph algorithms traverse networks and find shortest paths.",
},
];
}
/**
* Print search results with colored formatting.
*/
function printResults(query, results) {
if (results.length > 0) {
console.log(
chalk.cyan(`\nTop ${results.length} result(s) for '${query}':`)
);
results.forEach((r, i) => {
console.log(
` ${chalk.yellow(i + 1 + ".")} ${chalk.white(r.name)} ${chalk.gray(
"(score: " + r.score.toFixed(4) + ")"
)}`
);
});
} else {
console.log(chalk.red(`No results found for '${query}'.`));
}
}
/**
* Run interactive search mode.
*/
async function interactiveMode(engine, topK) {
const readline = require("readline");
const rl = readline.createInterface({
input: process.stdin,
output: process.stdout,
});
console.log(
chalk.cyan("Interactive search mode. Type 'quit' or 'exit' to stop.\n")
);
const askQuery = () => {
rl.question(chalk.green("Query> "), (line) => {
const trimmed = line.trim();
if (
trimmed.toLowerCase() === "quit" ||
trimmed.toLowerCase() === "exit" ||
trimmed.toLowerCase() === "q"
) {
console.log("Exiting.");
rl.close();
return;
}
if (trimmed.length === 0) {
askQuery();
return;
}
const results = engine.search(trimmed, topK);
printResults(trimmed, results);
console.log();
askQuery();
});
};
askQuery();
}
/**
* Parse command-line arguments.
*/
function parseArgs(argv) {
const args = {
dir: null,
query: null,
topK: 5,
interactive: false,
demo: false,
noStemming: false,
};
for (let i = 2; i < argv.length; i++) {
switch (argv[i]) {
case "--dir":
args.dir = argv[++i];
break;
case "--query":
case "-q":
args.query = argv[++i];
break;
case "--top-k":
case "-k":
args.topK = parseInt(argv[++i], 10);
break;
case "--interactive":
case "-i":
args.interactive = true;
break;
case "--demo":
args.demo = true;
break;
case "--no-stemming":
args.noStemming = true;
break;
default:
console.error(chalk.red(`Unknown argument: ${argv[i]}`));
process.exit(1);
}
}
return args;
}
// --- Main ---
function main() {
const args = parseArgs(process.argv);
const engine = new TFIDFSearchEngine({ useStemming: !args.noStemming });
// Load documents
if (args.dir) {
engine.loadFromDirectory(args.dir);
} else if (args.demo || (!args.query && !args.interactive)) {
console.log(chalk.cyan("Running with sample documents...\n"));
const samples = createSampleDocuments();
for (const doc of samples) {
engine.addDocument(doc.name, doc.content);
}
} else {
console.error(
chalk.red("No document source specified. Use --dir or --demo.")
);
process.exit(1);
}
console.log(
chalk.green(`Index built: ${engine.getDocumentCount()} documents.\n`)
);
// Execute search
if (args.interactive) {
interactiveMode(engine, args.topK);
} else if (args.query) {
const results = engine.search(args.query, args.topK);
printResults(args.query, results);
} else {
// Default demo queries
const sampleQueries = [
"machine learning neural networks",
"python programming language",
"web frontend JavaScript",
"sorting search algorithms",
];
for (const q of sampleQueries) {
const results = engine.search(q, 3);
printResults(q, results);
console.log();
}
}
}
main();