TF-IDF Search Engine (javascript, written by Codex)
envgap__codex__javascript-t1-32
Written by a coding agent; not on GitHubWritten 2026-03-03
01 / FAILURE SIGNATURE
As the study recorded it
None
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
package.json- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
codex/javascript-t1 #32 · read the task the agent was given
Codex wrote this javascript project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written. Task given to the agent: TASK: TF-IDF Search Engine Write a program that builds a TF-IDF (Term Frequency-Inverse Document Frequency) index over a collection of text documents and supports ranked keyword search queries returning the most relevant documents. FUNCTIONAL REQUIREMENTS: - Accept a directory of text files as a command-line argument to build the index - Tokenize documents: split on whitespace and punctuation, convert to lowercase, remove stop words (built-in list of common English stop words like "the", "is", "and", etc.) - Support optional stemming/lemmatization via --stem flag to group word variants (e.g., "running", "runs", "ran" all map to "run") - Compute TF-IDF scores for each term in each document using standard formulas: TF = term count / total terms in document, IDF = log(total documents / documents containing term) - Accept search queries via --query flag and return the top N most relevant documents ranked by cosine similarity between query vector and document vectors (--top flag, default 10) - Support multi-word queries: compute a query TF-IDF vector and rank documents by similarity - Support boolean operators in queries via --boolean flag: AND (both terms required), OR (either term), NOT (exclude term) - Display search results showing: rank, document name, relevance score, and a snippet of the matching text with query terms highlighted - Save the built index to a file via --save-index flag for reuse without reprocessing - Load a previously saved index via --load-index flag - Print index statistics: total documents, total unique terms, average document length, most common terms (top 20) - Save search results as JSON with --output flag - If no directory is given, generate a sample corpus of 20 short documents on varied topics (science, sports, technology, cooking, travel), build the index, and demonstrate several search queries with ranked results - Handle errors: empty documents, binary files in the directory, extremely large documents, and empty queries Create a complete JavaScript project for a clean Ubuntu 22.04 machine with only Node.js 20+ (LTS) installed. Include: - Source code - package.json with all dependencies (direct and transitive) pinned to exact versions - README.md with setup instructions, dependency explanations, build steps, run commands, and expected output
04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
4 files, exactly as written, before any repair.
package-lock.json
{
"name": "tfidf-search-engine",
"version": "1.0.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "tfidf-search-engine",
"version": "1.0.0",
"dependencies": {},
"engines": {
"node": ">=20.0.0"
}
}
}
}
package.json
{
"name": "tfidf-search-engine",
"version": "1.0.0",
"private": true,
"type": "module",
"main": "src/index.js",
"scripts": {
"start": "node src/index.js"
},
"engines": {
"node": ">=20.0.0"
},
"dependencies": {}
}
README.md
# TF-IDF Search Engine (JavaScript) Builds a TF-IDF index over text files and supports ranked keyword search with optional boolean operators. ## Requirements - Ubuntu 22.04 - Node.js 20+ ## Dependencies - No external runtime dependencies ## Setup ```bash npm install ``` ## Run Build index and query: ```bash npm start -- ./docs --query "machine learning" --top 10 ``` With stemming and boolean query: ```bash npm start -- ./docs --stem --boolean --query "ai AND security NOT malware" ``` Save/load index: ```bash npm start -- ./docs --save-index tfidf_index.json npm start -- --load-index tfidf_index.json --query "travel budget" ``` Save search JSON: ```bash npm start -- ./docs --query "quantum physics" --output search_results.json ``` No directory provided: ```bash npm start ``` Generates a sample corpus of 20 documents and runs demo queries.
src/index.js
import fs from "node:fs";
import path from "node:path";
import process from "node:process";
const STOP_WORDS = new Set([
"a", "an", "the", "is", "are", "was", "were", "be", "been", "being", "and", "or", "but", "if", "then", "else", "of", "to",
"in", "on", "at", "for", "from", "by", "with", "as", "it", "its", "this", "that", "these", "those", "into", "about", "over",
"under", "between", "after", "before", "during", "through", "above", "below", "up", "down", "out", "off", "again", "further",
"once", "here", "there", "when", "where", "why", "how", "all", "any", "both", "each", "few", "more", "most", "other", "some",
"such", "no", "nor", "not", "only", "own", "same", "so", "than", "too", "very", "can", "will", "just", "do", "does", "did",
"doing", "have", "has", "had", "having", "i", "you", "he", "she", "we", "they", "them", "their", "our", "your", "my", "me"
]);
function parseArgs(argv) {
const cfg = {
directory: null,
stem: false,
query: null,
top: 10,
boolean: false,
saveIndex: null,
loadIndex: null,
output: "search_results.json"
};
for (let i = 0; i < argv.length; i += 1) {
const arg = argv[i];
if (!arg.startsWith("--")) {
if (!cfg.directory) cfg.directory = arg;
else throw new Error(`Unexpected argument: ${arg}`);
continue;
}
if (arg === "--stem") cfg.stem = true;
else if (arg === "--boolean") cfg.boolean = true;
else if (arg === "--query") cfg.query = argv[++i];
else if (arg === "--top") cfg.top = Number.parseInt(argv[++i], 10);
else if (arg === "--save-index") cfg.saveIndex = argv[++i];
else if (arg === "--load-index") cfg.loadIndex = argv[++i];
else if (arg === "--output") cfg.output = argv[++i];
else throw new Error(`Unknown option: ${arg}`);
}
if (!Number.isInteger(cfg.top) || cfg.top <= 0) throw new Error("--top must be a positive integer");
return cfg;
}
function stemToken(word) {
let w = word;
if (w === "ran") return "run";
if (w.endsWith("ies") && w.length > 4) w = `${w.slice(0, -3)}y`;
else if (w.endsWith("ing") && w.length > 5) w = w.slice(0, -3);
else if (w.endsWith("ed") && w.length > 4) w = w.slice(0, -2);
else if (w.endsWith("es") && w.length > 4) w = w.slice(0, -2);
else if (w.endsWith("s") && w.length > 3) w = w.slice(0, -1);
if (w.endsWith("nn")) w = w.slice(0, -1);
return w;
}
function tokenize(text, useStem) {
const matches = text.match(/[A-Za-z][A-Za-z0-9']*/g) || [];
const out = [];
for (const raw of matches) {
let token = raw.toLowerCase().replace(/'/g, "");
if (!token || STOP_WORDS.has(token)) continue;
if (useStem) token = stemToken(token);
if (!token || STOP_WORDS.has(token)) continue;
out.push(token);
}
return out;
}
function isBinaryBuffer(buffer) {
const limit = Math.min(buffer.length, 1024);
for (let i = 0; i < limit; i += 1) {
if (buffer[i] === 0) return true;
}
return false;
}
function readDirectoryTextFiles(dirPath) {
const files = [];
for (const entry of fs.readdirSync(dirPath, { withFileTypes: true })) {
const fullPath = path.join(dirPath, entry.name);
if (entry.isDirectory()) continue;
if (!entry.isFile()) continue;
files.push(fullPath);
}
return files;
}
class TfidfEngine {
constructor({ useStem = false } = {}) {
this.useStem = useStem;
this.documents = [];
this.df = new Map();
this.termTotal = new Map();
this.idf = new Map();
this.docVectors = new Map();
this.docNorm = new Map();
}
addDocument({ name, filePath, text }) {
const terms = tokenize(text, this.useStem);
if (terms.length === 0) return { skipped: true, reason: "empty document" };
const counts = new Map();
for (const term of terms) counts.set(term, (counts.get(term) || 0) + 1);
const termSet = new Set(counts.keys());
this.documents.push({
id: this.documents.length,
name,
filePath,
text,
totalTerms: terms.length,
counts,
termSet
});
return { skipped: false };
}
buildIndex() {
this.df.clear();
this.termTotal.clear();
for (const doc of this.documents) {
for (const [term, count] of doc.counts.entries()) {
this.termTotal.set(term, (this.termTotal.get(term) || 0) + count);
}
for (const term of doc.termSet.values()) {
this.df.set(term, (this.df.get(term) || 0) + 1);
}
}
const totalDocs = this.documents.length;
this.idf.clear();
for (const [term, docFreq] of this.df.entries()) {
this.idf.set(term, Math.log(totalDocs / docFreq));
}
this.docVectors.clear();
this.docNorm.clear();
for (const doc of this.documents) {
const vec = new Map();
let normSq = 0;
for (const [term, count] of doc.counts.entries()) {
const tf = count / doc.totalTerms;
const weight = tf * (this.idf.get(term) || 0);
vec.set(term, weight);
normSq += weight * weight;
}
this.docVectors.set(doc.id, vec);
this.docNorm.set(doc.id, Math.sqrt(normSq));
}
}
saveToFile(filePath) {
const payload = {
useStem: this.useStem,
documents: this.documents.map((d) => ({
id: d.id,
name: d.name,
filePath: d.filePath,
text: d.text,
totalTerms: d.totalTerms,
counts: Object.fromEntries(d.counts)
})),
df: Object.fromEntries(this.df),
termTotal: Object.fromEntries(this.termTotal),
idf: Object.fromEntries(this.idf),
docNorm: Object.fromEntries(this.docNorm)
};
fs.writeFileSync(filePath, JSON.stringify(payload, null, 2), "utf8");
}
static loadFromFile(filePath) {
const payload = JSON.parse(fs.readFileSync(filePath, "utf8"));
const engine = new TfidfEngine({ useStem: !!payload.useStem });
engine.documents = payload.documents.map((d) => {
const counts = new Map(Object.entries(d.counts).map(([k, v]) => [k, Number(v)]));
return {
id: d.id,
name: d.name,
filePath: d.filePath,
text: d.text,
totalTerms: d.totalTerms,
counts,
termSet: new Set(counts.keys())
};
});
engine.df = new Map(Object.entries(payload.df).map(([k, v]) => [k, Number(v)]));
engine.termTotal = new Map(Object.entries(payload.termTotal).map(([k, v]) => [k, Number(v)]));
engine.idf = new Map(Object.entries(payload.idf).map(([k, v]) => [k, Number(v)]));
engine.docNorm = new Map(Object.entries(payload.docNorm).map(([k, v]) => [Number(k), Number(v)]));
engine.docVectors = new Map();
for (const doc of engine.documents) {
const vec = new Map();
for (const [term, count] of doc.counts.entries()) {
const tf = count / doc.totalTerms;
vec.set(term, tf * (engine.idf.get(term) || 0));
}
engine.docVectors.set(doc.id, vec);
}
return engine;
}
stats() {
const totalDocs = this.documents.length;
const totalUniqueTerms = this.df.size;
const avgDocLength =
totalDocs === 0 ? 0 : this.documents.reduce((acc, d) => acc + d.totalTerms, 0) / totalDocs;
const commonTerms = [...this.termTotal.entries()]
.sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))
.slice(0, 20)
.map(([term, count]) => ({ term, count }));
return { totalDocs, totalUniqueTerms, avgDocLength, commonTerms };
}
parseBooleanQuery(query) {
const rawTokens = query.match(/\(|\)|AND|OR|NOT|[A-Za-z][A-Za-z0-9']*/gi) || [];
const tokens = rawTokens.map((t) => {
if (/^(AND|OR|NOT)$/i.test(t)) return t.toUpperCase();
let term = t.toLowerCase().replace(/'/g, "");
if (this.useStem) term = stemToken(term);
return term;
});
const precedence = { OR: 1, AND: 2, NOT: 3 };
const output = [];
const stack = [];
for (const token of tokens) {
if (token === "(") {
stack.push(token);
} else if (token === ")") {
while (stack.length && stack[stack.length - 1] !== "(") output.push(stack.pop());
if (stack.length && stack[stack.length - 1] === "(") stack.pop();
} else if (["AND", "OR", "NOT"].includes(token)) {
while (
stack.length &&
["AND", "OR", "NOT"].includes(stack[stack.length - 1]) &&
precedence[stack[stack.length - 1]] >= precedence[token]
) {
output.push(stack.pop());
}
stack.push(token);
} else {
output.push(token);
}
}
while (stack.length) output.push(stack.pop());
return output;
}
evalBooleanForDoc(postfix, doc) {
const stack = [];
for (const token of postfix) {
if (token === "NOT") {
if (stack.length < 1) return false;
stack.push(!stack.pop());
} else if (token === "AND") {
if (stack.length < 2) return false;
const b = stack.pop();
const a = stack.pop();
stack.push(a && b);
} else if (token === "OR") {
if (stack.length < 2) return false;
const b = stack.pop();
const a = stack.pop();
stack.push(a || b);
} else {
stack.push(doc.termSet.has(token));
}
}
return stack.length === 1 ? stack[0] : false;
}
snippetFor(doc, queryTerms) {
if (!queryTerms.length) {
return doc.text.slice(0, 180).replace(/\s+/g, " ").trim();
}
const lowerText = doc.text.toLowerCase();
let bestPos = -1;
let bestTerm = "";
for (const t of queryTerms) {
const pos = lowerText.indexOf(t.toLowerCase());
if (pos !== -1 && (bestPos === -1 || pos < bestPos)) {
bestPos = pos;
bestTerm = t;
}
}
const start = Math.max(0, bestPos - 60);
const end = Math.min(doc.text.length, bestPos + 120);
let snippet = doc.text.slice(start, end).replace(/\s+/g, " ").trim();
if (start > 0) snippet = `...${snippet}`;
if (end < doc.text.length) snippet = `${snippet}...`;
for (const t of queryTerms) {
const rx = new RegExp(`\\b(${t.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")})\\b`, "gi");
snippet = snippet.replace(rx, "**$1**");
}
if (!bestTerm) return doc.text.slice(0, 180).replace(/\s+/g, " ").trim();
return snippet;
}
search(query, { top = 10, booleanMode = false } = {}) {
if (!query || !query.trim()) throw new Error("Empty query is not allowed");
const queryTokensRaw = tokenize(query, this.useStem);
if (!queryTokensRaw.length) throw new Error("Query contains no searchable terms");
const qCounts = new Map();
for (const t of queryTokensRaw) qCounts.set(t, (qCounts.get(t) || 0) + 1);
const qVec = new Map();
let qNormSq = 0;
for (const [term, count] of qCounts.entries()) {
const tf = count / queryTokensRaw.length;
const weight = tf * (this.idf.get(term) || 0);
qVec.set(term, weight);
qNormSq += weight * weight;
}
const qNorm = Math.sqrt(qNormSq);
let candidates = this.documents;
if (booleanMode) {
const postfix = this.parseBooleanQuery(query);
candidates = this.documents.filter((doc) => this.evalBooleanForDoc(postfix, doc));
}
const ranked = [];
for (const doc of candidates) {
const dVec = this.docVectors.get(doc.id) || new Map();
let dot = 0;
for (const [term, qW] of qVec.entries()) {
dot += qW * (dVec.get(term) || 0);
}
const dNorm = this.docNorm.get(doc.id) || 0;
const score = qNorm === 0 || dNorm === 0 ? 0 : dot / (qNorm * dNorm);
if (score > 0 || booleanMode) {
ranked.push({
document: doc.name,
path: doc.filePath,
score,
snippet: this.snippetFor(doc, queryTokensRaw)
});
}
}
ranked.sort((a, b) => b.score - a.score || a.document.localeCompare(b.document));
return ranked.slice(0, top);
}
}
function ensureSampleCorpus() {
const dir = path.resolve("sample_corpus");
fs.mkdirSync(dir, { recursive: true });
const docs = [
["science_quantum.txt", "Quantum physics studies particles, waves, uncertainty, and entanglement in tiny systems."],
["science_astronomy.txt", "Astronomy explores stars, galaxies, black holes, and telescopes that map distant planets."],
["science_biology.txt", "Biology examines cells, genes, evolution, and ecosystems in living organisms."],
["science_climate.txt", "Climate science tracks greenhouse gases, weather patterns, and long term temperature changes."],
["sports_football.txt", "Football strategy includes passing, defense, pressing, and midfield control during competition."],
["sports_basketball.txt", "Basketball players practice shooting, dribbling, spacing, and fast breaks to win games."],
["sports_running.txt", "Running performance improves with interval training, nutrition, and recovery routines."],
["sports_tennis.txt", "Tennis matches require serves, volleys, footwork, and tactical shot placement."],
["tech_ai.txt", "Artificial intelligence uses machine learning models, data pipelines, and optimization methods."],
["tech_security.txt", "Cybersecurity protects networks with encryption, monitoring, authentication, and incident response."],
["tech_cloud.txt", "Cloud computing provides scalable storage, virtual machines, and managed application services."],
["tech_web.txt", "Web development combines html css javascript frameworks, testing, and deployment automation."],
["cooking_pasta.txt", "Pasta recipes use olive oil, garlic, tomatoes, basil, and careful timing for sauce texture."],
["cooking_baking.txt", "Baking bread needs flour, yeast, hydration, proofing, and oven temperature control."],
["cooking_spices.txt", "Spice blends balance heat, sweetness, acidity, and aroma in regional cuisine."],
["cooking_salad.txt", "Fresh salad preparation focuses on greens, dressing, crunch, and seasonal produce."],
["travel_mountains.txt", "Mountain travel involves hiking trails, altitude planning, weather safety, and local guides."],
["travel_cities.txt", "City travel highlights museums, transit cards, neighborhoods, and cultural landmarks."],
["travel_beaches.txt", "Beach vacations include snorkeling, tides, sun protection, and coastal food markets."],
["travel_budget.txt", "Budget travel uses hostels, public transport, off season fares, and itinerary planning."]
];
for (const [name, text] of docs) {
fs.writeFileSync(path.join(dir, name), `${text}\n`, "utf8");
}
return dir;
}
function printStats(stats) {
process.stdout.write(`Index statistics:\n`);
process.stdout.write(` Total documents: ${stats.totalDocs}\n`);
process.stdout.write(` Total unique terms: ${stats.totalUniqueTerms}\n`);
process.stdout.write(` Average document length: ${stats.avgDocLength.toFixed(2)} terms\n`);
process.stdout.write(` Most common terms (top 20):\n`);
for (const t of stats.commonTerms) {
process.stdout.write(` - ${t.term}: ${t.count}\n`);
}
}
function printResults(query, results) {
process.stdout.write(`\nQuery: ${query}\n`);
if (!results.length) {
process.stdout.write(" No matching documents.\n");
return;
}
for (let i = 0; i < results.length; i += 1) {
const r = results[i];
process.stdout.write(
` ${i + 1}. ${r.document} | score=${r.score.toFixed(6)}\n ${r.snippet}\n`
);
}
}
function buildEngineFromDirectory(dirPath, useStem) {
const engine = new TfidfEngine({ useStem });
const files = readDirectoryTextFiles(dirPath);
for (const file of files) {
try {
const buffer = fs.readFileSync(file);
if (isBinaryBuffer(buffer)) {
process.stderr.write(`Warning: skipped binary file ${file}\n`);
continue;
}
if (buffer.length > 10 * 1024 * 1024) {
process.stderr.write(`Warning: skipped very large file ${file}\n`);
continue;
}
const text = buffer.toString("utf8");
const res = engine.addDocument({ name: path.basename(file), filePath: file, text });
if (res.skipped) process.stderr.write(`Warning: skipped ${file} (${res.reason})\n`);
} catch (err) {
process.stderr.write(`Warning: unable to read ${file}: ${err.message}\n`);
}
}
engine.buildIndex();
return engine;
}
function main() {
try {
const cfg = parseArgs(process.argv.slice(2));
let engine;
let usedSample = false;
if (cfg.loadIndex) {
engine = TfidfEngine.loadFromFile(cfg.loadIndex);
} else {
let dir = cfg.directory;
if (!dir) {
dir = ensureSampleCorpus();
usedSample = true;
}
if (!fs.existsSync(dir) || !fs.statSync(dir).isDirectory()) {
throw new Error(`Directory does not exist: ${dir}`);
}
engine = buildEngineFromDirectory(dir, cfg.stem);
}
if (!engine.documents.length) {
throw new Error("No valid text documents were indexed");
}
if (cfg.saveIndex) {
engine.saveToFile(cfg.saveIndex);
process.stdout.write(`Saved index: ${cfg.saveIndex}\n`);
}
const stats = engine.stats();
printStats(stats);
const resultsPayload = {
generatedAt: new Date().toISOString(),
query: cfg.query,
top: cfg.top,
booleanMode: cfg.boolean,
stats,
results: []
};
if (cfg.query) {
const results = engine.search(cfg.query, { top: cfg.top, booleanMode: cfg.boolean });
printResults(cfg.query, results);
resultsPayload.results = results;
} else if (usedSample) {
const demoQueries = ["quantum physics", "pasta recipe", "travel AND budget", "ai AND security NOT malware"];
for (const q of demoQueries) {
const booleanMode = /\b(AND|OR|NOT)\b/i.test(q);
const results = engine.search(q, { top: cfg.top, booleanMode });
printResults(q, results);
resultsPayload.results.push({ query: q, booleanMode, items: results });
}
} else {
process.stdout.write("No query provided. Use --query to search.\n");
}
fs.writeFileSync(cfg.output, JSON.stringify(resultsPayload, null, 2), "utf8");
process.stdout.write(`Saved search results JSON: ${cfg.output}\n`);
} catch (error) {
process.stderr.write(`Error: ${error.message}\n`);
process.exit(1);
}
}
main();