Text Diff Tool (python, written by Codex)
envgap__codex__python-t1-33
Written by a coding agent; not on GitHubWritten 2026-03-03
01 / FAILURE SIGNATURE
As the study recorded it
None
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
requirements.txt- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
codex/python-t1 #33 · read the task the agent was given
Codex wrote this python project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written. Task given to the agent: TASK: Text Diff Tool Write a program that compares two text files and produces a detailed diff showing additions, deletions, and modifications with configurable output formats and context control. FUNCTIONAL REQUIREMENTS: - Accept two file paths as command-line arguments (original and modified) - Compute the longest common subsequence (LCS) based diff to identify added, deleted, and changed lines - Support multiple output formats via --format flag: unified diff (default, similar to git diff), side-by-side (two-column view), inline (changes marked within lines), and html (visual diff as an HTML page) - Support configurable context lines around changes via --context flag (default: 3 lines of unchanged context around each change) - Detect and highlight intra-line changes: when a line is modified, show exactly which words or characters changed within the line - Support ignoring whitespace differences via --ignore-whitespace flag - Support ignoring case differences via --ignore-case flag - Support ignoring blank lines via --ignore-blank-lines flag - Compute and display diff statistics: total lines in each file, lines added, lines deleted, lines modified, and a similarity percentage - Support comparing directories via --recursive flag: compare all matching files in two directory trees and report which files are added, deleted, modified, or identical - Apply color coding in console output: green for additions, red for deletions, yellow for modifications - Save the diff output to a file via --output flag - If no input files are given, generate two sample text files (original and modified version with insertions, deletions, modifications, and moved blocks), then compute and display the diff in all supported formats - Handle errors: binary files (detect and skip with warning), missing files, encoding mismatches, and very large files Create a complete Python project for a clean Ubuntu 22.04 machine with only Python 3.10+ installed. Include: - Source code - requirements.txt with all dependencies (direct and transitive) pinned to exact versions - README.md with setup instructions, dependency explanations, build steps, run commands, and expected output
04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
3 files, exactly as written, before any repair.
README.md
# Text Diff Tool (Python) Compares two text files using an LCS-based diff and outputs unified, side-by-side, inline, or HTML formats. ## Requirements - Ubuntu 22.04 - Python 3.10+ ## Dependencies - No external dependencies (standard library only) ## Setup ```bash python3 -m venv .venv source .venv/bin/activate pip install -r requirements.txt ``` ## Run Unified diff: ```bash python3 src/main.py original.txt modified.txt --format unified --context 3 ``` Inline diff with ignores: ```bash python3 src/main.py original.txt modified.txt --format inline --ignore-whitespace --ignore-case ``` HTML diff: ```bash python3 src/main.py original.txt modified.txt --format html --output diff.html ``` Recursive directory comparison: ```bash python3 src/main.py dir_a dir_b --recursive ``` No input files: ```bash python3 src/main.py ``` Generates sample files and displays all supported formats.
requirements.txt
# No external dependencies required. # Uses Python 3.10+ standard library only.
src/main.py
#!/usr/bin/env python3
"""Text diff tool with LCS-based line diff and multiple output formats."""
from __future__ import annotations
import argparse
import json
import re
from dataclasses import dataclass
from pathlib import Path
ANSI = {"reset": "\x1b[0m", "red": "\x1b[31m", "green": "\x1b[32m", "yellow": "\x1b[33m"}
def colorize(text: str, color: str) -> str:
return f"{ANSI.get(color, '')}{text}{ANSI['reset']}"
@dataclass
class Config:
format: str
context: int
ignore_whitespace: bool
ignore_case: bool
ignore_blank_lines: bool
recursive: bool
output: str | None
inputs: list[str]
def parse_args() -> Config:
parser = argparse.ArgumentParser(description="Text Diff Tool")
parser.add_argument("inputs", nargs="*")
parser.add_argument("--format", default="unified", choices=["unified", "side-by-side", "inline", "html"])
parser.add_argument("--context", type=int, default=3)
parser.add_argument("--ignore-whitespace", action="store_true")
parser.add_argument("--ignore-case", action="store_true")
parser.add_argument("--ignore-blank-lines", action="store_true")
parser.add_argument("--recursive", action="store_true")
parser.add_argument("--output")
args = parser.parse_args()
if args.context < 0:
raise ValueError("--context must be a non-negative integer")
return Config(
format=args.format,
context=args.context,
ignore_whitespace=args.ignore_whitespace,
ignore_case=args.ignore_case,
ignore_blank_lines=args.ignore_blank_lines,
recursive=args.recursive,
output=args.output,
inputs=args.inputs,
)
def maybe_normalize(line: str, cfg: Config) -> str:
x = line
if cfg.ignore_whitespace:
x = re.sub(r"\s+", " ", x).strip()
if cfg.ignore_case:
x = x.lower()
return x
def is_binary(data: bytes) -> bool:
return b"\x00" in data[:1024]
def read_lines(path: Path, cfg: Config) -> tuple[bool, list[str], list[str]]:
data = path.read_bytes()
if is_binary(data):
return (True, [], [])
if len(data) > 20 * 1024 * 1024:
raise ValueError(f"Very large file skipped: {path}")
text = data.decode("utf-8", errors="replace")
lines = text.replace("\r\n", "\n").split("\n")
if cfg.ignore_blank_lines:
lines = [l for l in lines if l.strip() != ""]
norm = [maybe_normalize(l, cfg) for l in lines]
return (False, lines, norm)
def lcs_diff(a: list[str], b: list[str]) -> list[dict]:
n, m = len(a), len(b)
dp = [[0] * (m + 1) for _ in range(n + 1)]
for i in range(n - 1, -1, -1):
for j in range(m - 1, -1, -1):
dp[i][j] = dp[i + 1][j + 1] + 1 if a[i] == b[j] else max(dp[i + 1][j], dp[i][j + 1])
ops = []
i = j = 0
while i < n and j < m:
if a[i] == b[j]:
ops.append({"type": "equal", "ai": i, "bj": j})
i += 1
j += 1
elif dp[i + 1][j] >= dp[i][j + 1]:
ops.append({"type": "del", "ai": i, "bj": None})
i += 1
else:
ops.append({"type": "add", "ai": None, "bj": j})
j += 1
while i < n:
ops.append({"type": "del", "ai": i, "bj": None})
i += 1
while j < m:
ops.append({"type": "add", "ai": None, "bj": j})
j += 1
return ops
def pair_modifications(ops: list[dict]) -> list[dict]:
out = []
i = 0
while i < len(ops):
if ops[i]["type"] != "del":
out.append(ops[i])
i += 1
continue
di = i
while di < len(ops) and ops[di]["type"] == "del":
di += 1
ai = di
while ai < len(ops) and ops[ai]["type"] == "add":
ai += 1
if di < ai:
dels = ops[i:di]
adds = ops[di:ai]
k = min(len(dels), len(adds))
for t in range(k):
out.append({"type": "mod", "ai": dels[t]["ai"], "bj": adds[t]["bj"]})
out.extend(dels[k:])
out.extend(adds[k:])
i = ai
else:
out.append(ops[i])
i += 1
return out
def tokenize_inline(line: str) -> list[str]:
return [x for x in re.split(r"(\s+|[^\w\s]+)", line) if x != ""]
def inline_word_diff(old: str, new: str) -> tuple[str, str]:
a = tokenize_inline(old)
b = tokenize_inline(new)
ops = lcs_diff(a, b)
out_a, out_b = [], []
for op in ops:
if op["type"] == "equal":
out_a.append(a[op["ai"]])
out_b.append(b[op["bj"]])
elif op["type"] == "del":
out_a.append(f"[-{a[op['ai']]}-]")
elif op["type"] == "add":
out_b.append(f"[+{b[op['bj']]}+]")
return ("".join(out_a), "".join(out_b))
def diff_stats(ops: list[dict], a_len: int, b_len: int) -> dict:
added = sum(1 for x in ops if x["type"] == "add")
deleted = sum(1 for x in ops if x["type"] == "del")
modified = sum(1 for x in ops if x["type"] == "mod")
equal = sum(1 for x in ops if x["type"] == "equal")
similarity = 100.0 if max(a_len, b_len) == 0 else (equal / max(a_len, b_len)) * 100.0
return {
"total_original": a_len,
"total_modified": b_len,
"added": added,
"deleted": deleted,
"modified": modified,
"similarity_pct": similarity,
}
def render_unified(ops: list[dict], left: list[str], right: list[str], context: int) -> str:
changed = [i for i, op in enumerate(ops) if op["type"] != "equal"]
if not changed:
return "No differences.\n"
keep = set()
for idx in changed:
for k in range(max(0, idx - context), min(len(ops), idx + context + 1)):
keep.add(k)
lines = ["--- original", "+++ modified"]
skipping = False
for i, op in enumerate(ops):
if i not in keep:
if not skipping:
lines.append("@@ ... @@")
skipping = True
continue
skipping = False
t = op["type"]
if t == "equal":
lines.append(" " + left[op["ai"]])
elif t == "add":
lines.append(colorize("+" + right[op["bj"]], "green"))
elif t == "del":
lines.append(colorize("-" + left[op["ai"]], "red"))
else:
oa, nb = inline_word_diff(left[op["ai"]], right[op["bj"]])
lines.append(colorize("~" + oa, "yellow"))
lines.append(colorize("~" + nb, "yellow"))
return "\n".join(lines) + "\n"
def render_side_by_side(ops: list[dict], left: list[str], right: list[str]) -> str:
out = []
width = 60
for op in ops:
t = op["type"]
l = "" if op["ai"] is None else left[op["ai"]]
r = "" if op["bj"] is None else right[op["bj"]]
if t == "add":
out.append(colorize(f"{'':{width}} | + {r}", "green"))
elif t == "del":
out.append(colorize(f"{l:{width}} | - ", "red"))
elif t == "mod":
oa, nb = inline_word_diff(l, r)
out.append(colorize(f"{oa:{width}} | ~ {nb}", "yellow"))
else:
out.append(f"{l:{width}} | {r}")
return "\n".join(out) + "\n"
def render_inline(ops: list[dict], left: list[str], right: list[str]) -> str:
out = []
for op in ops:
t = op["type"]
if t == "equal":
out.append(" " + left[op["ai"]])
elif t == "add":
out.append(colorize("+ " + right[op["bj"]], "green"))
elif t == "del":
out.append(colorize("- " + left[op["ai"]], "red"))
else:
oa, nb = inline_word_diff(left[op["ai"]], right[op["bj"]])
out.append(colorize("~ " + oa, "yellow"))
out.append(colorize("~ " + nb, "yellow"))
return "\n".join(out) + "\n"
def html_escape(s: str) -> str:
return s.replace("&", "&").replace("<", "<").replace(">", ">").replace('"', """).replace("'", "'")
def render_html(ops: list[dict], left: list[str], right: list[str]) -> str:
rows = []
for op in ops:
t = op["type"]
if t == "equal":
rows.append(f"<tr class='eq'><td>{html_escape(left[op['ai']])}</td><td>{html_escape(right[op['bj']])}</td></tr>")
elif t == "add":
rows.append(f"<tr class='add'><td></td><td>{html_escape(right[op['bj']])}</td></tr>")
elif t == "del":
rows.append(f"<tr class='del'><td>{html_escape(left[op['ai']])}</td><td></td></tr>")
else:
oa, nb = inline_word_diff(left[op["ai"]], right[op["bj"]])
rows.append(f"<tr class='mod'><td>{html_escape(oa)}</td><td>{html_escape(nb)}</td></tr>")
return (
"<!doctype html><html><head><meta charset='utf-8'><title>Diff</title>"
"<style>body{font-family:Arial,sans-serif;margin:1rem}table{width:100%;border-collapse:collapse}"
"td{border:1px solid #ddd;padding:.4rem;vertical-align:top;white-space:pre-wrap;font-family:Consolas,monospace}"
".add td{background:#e8ffe8}.del td{background:#ffe8e8}.mod td{background:#fff7db}</style></head>"
f"<body><h1>Diff</h1><table>{''.join(rows)}</table></body></html>\n"
)
def diff_two_files(left_path: Path, right_path: Path, cfg: Config) -> dict:
if not left_path.exists() or not right_path.exists():
raise FileNotFoundError("Missing input files")
left_bin, left_lines, left_norm = read_lines(left_path, cfg)
right_bin, right_lines, right_norm = read_lines(right_path, cfg)
if left_bin or right_bin:
return {
"warning": "Binary file detected; diff skipped",
"output": "",
"stats": {"total_original": 0, "total_modified": 0, "added": 0, "deleted": 0, "modified": 0, "similarity_pct": 0.0},
"ops": [],
}
ops = pair_modifications(lcs_diff(left_norm, right_norm))
stats = diff_stats(ops, len(left_lines), len(right_lines))
if cfg.format == "unified":
output = render_unified(ops, left_lines, right_lines, cfg.context)
elif cfg.format == "side-by-side":
output = render_side_by_side(ops, left_lines, right_lines)
elif cfg.format == "inline":
output = render_inline(ops, left_lines, right_lines)
else:
output = render_html(ops, left_lines, right_lines)
return {"warning": None, "output": output, "stats": stats, "ops": ops}
def walk_files(root: Path) -> dict[str, Path]:
out: dict[str, Path] = {}
for p in root.rglob("*"):
if p.is_file():
out[str(p.relative_to(root))] = p
return out
def diff_directories(a: Path, b: Path, cfg: Config) -> dict:
files_a = walk_files(a)
files_b = walk_files(b)
all_rel = sorted(set(files_a.keys()) | set(files_b.keys()))
summary = {"added": [], "deleted": [], "modified": [], "identical": []}
for rel in all_rel:
pa = files_a.get(rel)
pb = files_b.get(rel)
if pa is None:
summary["added"].append(rel)
continue
if pb is None:
summary["deleted"].append(rel)
continue
res = diff_two_files(pa, pb, cfg)
if res["warning"] or any(op["type"] != "equal" for op in res["ops"]):
summary["modified"].append(rel)
else:
summary["identical"].append(rel)
lines = ["Directory diff summary", f"Added: {len(summary['added'])}"]
lines += [colorize(f" + {x}", "green") for x in summary["added"]]
lines += [f"Deleted: {len(summary['deleted'])}"]
lines += [colorize(f" - {x}", "red") for x in summary["deleted"]]
lines += [f"Modified: {len(summary['modified'])}"]
lines += [colorize(f" ~ {x}", "yellow") for x in summary["modified"]]
lines += [f"Identical: {len(summary['identical'])}"]
lines += [f" {x}" for x in summary["identical"]]
return {"summary": summary, "output": "\n".join(lines) + "\n"}
def create_samples() -> tuple[Path, Path]:
left = Path("sample_original.txt").resolve()
right = Path("sample_modified.txt").resolve()
left.write_text(
"\n".join(
[
"Project Delta Status Report",
"The team completed phase one on Monday.",
"We tested the API and database integration.",
"Performance baseline is 220 requests per second.",
"Risks include deployment timing and data migration.",
"Action: finalize rollback plan.",
]
)
+ "\n",
encoding="utf-8",
)
right.write_text(
"\n".join(
[
"Project Delta Status Report",
"The team completed phase one on Tuesday.",
"We tested API integration and caching layer.",
"Performance baseline is 260 requests per second.",
"Action: finalize rollback plan and run rehearsal.",
"New note: monitor latency in production.",
]
)
+ "\n",
encoding="utf-8",
)
return left, right
def write_output(path: str | None, content: str) -> None:
if not path:
return
Path(path).write_text(content, encoding="utf-8")
def main() -> int:
cfg = parse_args()
sample_mode = False
if len(cfg.inputs) >= 2:
left, right = Path(cfg.inputs[0]), Path(cfg.inputs[1])
else:
left, right = create_samples()
sample_mode = True
if cfg.recursive:
if not left.exists() or not left.is_dir():
raise ValueError("First input must be a directory for --recursive")
if not right.exists() or not right.is_dir():
raise ValueError("Second input must be a directory for --recursive")
res = diff_directories(left, right, cfg)
print(res["output"], end="")
write_output(cfg.output, res["output"])
return 0
if not left.exists() or not right.exists():
raise FileNotFoundError("Input files not found")
if left.is_dir() or right.is_dir():
raise ValueError("Use --recursive to compare directories")
formats = [cfg.format] if not sample_mode else ["unified", "side-by-side", "inline", "html"]
report_formats = []
written_blocks = []
for fmt in formats:
run_cfg = Config(
format=fmt,
context=cfg.context,
ignore_whitespace=cfg.ignore_whitespace,
ignore_case=cfg.ignore_case,
ignore_blank_lines=cfg.ignore_blank_lines,
recursive=cfg.recursive,
output=cfg.output,
inputs=cfg.inputs,
)
res = diff_two_files(left, right, run_cfg)
block = f"\n===== FORMAT: {fmt} =====\n{res['output']}" if sample_mode else res["output"]
if res["warning"]:
print(f"Warning: {res['warning']}")
print(block, end="")
if not sample_mode:
s = res["stats"]
print(
f"Stats: original={s['total_original']}, modified={s['total_modified']}, +{s['added']}, -{s['deleted']}, "
f"~{s['modified']}, similarity={s['similarity_pct']:.2f}%"
)
written_blocks.append(block)
report_formats.append({"format": fmt, "warning": res["warning"], "stats": res["stats"]})
if cfg.output:
write_output(cfg.output, "\n".join(written_blocks))
report = {
"generated_at": datetime_now_iso(),
"original": str(left),
"modified": str(right),
"recursive": False,
"formats": report_formats,
}
Path("diff_report.json").write_text(json.dumps(report, indent=2), encoding="utf-8")
return 0
def datetime_now_iso() -> str:
import datetime
return datetime.datetime.now(datetime.timezone.utc).isoformat()
if __name__ == "__main__":
raise SystemExit(main())