HTTP File Downloader (python, written by Codex)
envgap__codex__python-t1-24
Written by a coding agent; not on GitHubWritten 2026-03-03
01 / FAILURE SIGNATURE
As the study recorded it
None
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
requirements.txt- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
codex/python-t1 #24 · read the task the agent was given
Codex wrote this python project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written. Task given to the agent: TASK: HTTP File Downloader Write a program that downloads files from HTTP/HTTPS URLs with support for resumable downloads, progress tracking, concurrent downloads, and checksum verification. FUNCTIONAL REQUIREMENTS: - Accept one or more URLs as command-line arguments - Display a progress bar during download showing: percentage complete, bytes downloaded, total size, download speed, and estimated time remaining - Support resumable downloads via HTTP Range headers: if a download is interrupted, restarting with the same URL and output path should resume from where it stopped via --resume flag - Support concurrent downloading of multiple files via --parallel flag with configurable thread count (--threads, default 4) - Support downloading all URLs listed in a text file (one URL per line) via --list flag - Verify downloaded file integrity via --checksum flag accepting algorithm:hash format (e.g., --checksum sha256:abc123...) - Support custom HTTP headers via --header flag (e.g., --header "Authorization: Bearer token") - Support following HTTP redirects (up to 10 hops) and report the final URL - Set connection timeout via --timeout flag (default 30 seconds) and retry failed downloads via --retries flag (default 3) with exponential backoff - Save files to a directory specified by --output flag (default: current directory), using the filename from the URL or Content-Disposition header - Print a download summary to console: file name, size, time taken, average speed, and checksum verification result - If no URLs are given, download a set of sample public domain text files from Project Gutenberg, display progress for each, and print a summary table - Handle errors: DNS resolution failures, SSL certificate errors, HTTP 4xx/5xx responses, disk full, and network timeouts Create a complete Python project for a clean Ubuntu 22.04 machine with only Python 3.10+ installed. Include: - Source code - requirements.txt with all dependencies (direct and transitive) pinned to exact versions - README.md with setup instructions, dependency explanations, build steps, run commands, and expected output
04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
3 files, exactly as written, before any repair.
README.md
# HTTP File Downloader (Python) Downloads files from HTTP/HTTPS with resumable support, progress bars, redirects, retries, parallel downloads, header overrides, checksum verification, list-file input, and summary reporting. ## Requirements - Ubuntu 22.04 - Python 3.10+ ## Dependencies (Pinned) - `requests==2.32.3` - `tqdm==4.66.5` ## Setup ```bash python -m venv .venv source .venv/bin/activate pip install -r requirements.txt ``` ## Run ```bash python src/main.py https://example.com/file.txt python src/main.py https://example.com/file.bin --resume --output ./downloads python src/main.py --list urls.txt --parallel --threads 4 --timeout 30 --retries 3 python src/main.py https://example.com/file --checksum sha256:abc123... python src/main.py https://example.com/file --header "Authorization: Bearer TOKEN" python src/main.py ``` ## Notes - `--parallel` enables concurrent downloading (`--threads` default `4`). - `--resume` uses `Range` where server supports partial content. - `--checksum` verifies the first download item in multi-URL runs. - No-input mode downloads sample Project Gutenberg text files.
requirements.txt
requests==2.32.3 tqdm==4.66.5
src/main.py
#!/usr/bin/env python3
from __future__ import annotations
import argparse
import hashlib
import math
import os
import time
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
from urllib.parse import urlparse, unquote
import requests
from requests.exceptions import RequestException
from tqdm import tqdm
SAMPLE_URLS = [
"https://www.gutenberg.org/files/11/11-0.txt",
"https://www.gutenberg.org/files/1342/1342-0.txt",
"https://www.gutenberg.org/files/1661/1661-0.txt",
]
def parse_checksum(raw: str | None) -> tuple[str, str] | None:
if not raw:
return None
if ":" not in raw:
raise ValueError("Invalid --checksum. Use algorithm:hash.")
alg, expected = raw.split(":", 1)
return alg.lower().strip(), expected.lower().strip()
def format_bytes(n: float) -> str:
if n < 0:
return "?"
units = ["B", "KB", "MB", "GB", "TB"]
i = 0
while n >= 1024 and i < len(units) - 1:
n /= 1024.0
i += 1
return f"{n:.2f} {units[i]}" if i else f"{int(n)} {units[i]}"
def parse_header_pairs(pairs: list[str]) -> dict[str, str]:
out: dict[str, str] = {}
for p in pairs:
if ":" not in p:
continue
k, v = p.split(":", 1)
out[k.strip()] = v.strip()
return out
def filename_from_url(url: str) -> str:
parsed = urlparse(url)
name = Path(unquote(parsed.path)).name
return name if name else "download.bin"
def filename_from_content_disposition(cd: str | None) -> str | None:
if not cd:
return None
marker = "filename="
idx = cd.lower().find(marker)
if idx < 0:
return None
value = cd[idx + len(marker):].strip().strip('"')
return unquote(value) if value else None
def hash_file(path_value: Path, algorithm: str) -> str:
h = hashlib.new(algorithm)
with path_value.open("rb") as f:
while True:
chunk = f.read(1024 * 1024)
if not chunk:
break
h.update(chunk)
return h.hexdigest().lower()
def download_one(
url: str,
idx: int,
total_count: int,
args: argparse.Namespace,
checksum: tuple[str, str] | None,
) -> dict:
out_dir = Path(args.output or ".").resolve()
out_dir.mkdir(parents=True, exist_ok=True)
timeout = float(args.timeout or 30)
retries = int(args.retries or 3)
base_name = filename_from_url(url)
guess_path = out_dir / base_name
session = requests.Session()
session.max_redirects = 10
common_headers = parse_header_pairs(args.header or [])
existing = guess_path.stat().st_size if args.resume and guess_path.exists() else 0
attempt = 0
while attempt <= retries:
started = time.perf_counter()
headers = dict(common_headers)
if existing > 0:
headers["Range"] = f"bytes={existing}-"
try:
with session.get(url, headers=headers, stream=True, timeout=timeout, allow_redirects=True) as r:
if r.status_code >= 400:
raise RequestException(f"HTTP {r.status_code}")
final_url = r.url
file_name = filename_from_content_disposition(r.headers.get("Content-Disposition")) or filename_from_url(final_url)
out_path = out_dir / file_name
is_resume = bool(args.resume and existing > 0 and r.status_code == 206 and out_path == guess_path)
mode = "ab" if is_resume else "wb"
downloaded = existing if is_resume else 0
content_len = int(r.headers.get("Content-Length", "0") or 0)
total_size = content_len + downloaded if content_len > 0 else None
with out_path.open(mode) as f, tqdm(
total=total_size,
initial=downloaded,
unit="B",
unit_scale=True,
desc=f"[{idx + 1}/{total_count}] {out_path.name}",
leave=True,
) as bar:
for chunk in r.iter_content(chunk_size=1024 * 128):
if not chunk:
continue
f.write(chunk)
downloaded += len(chunk)
bar.update(len(chunk))
elapsed = max(0.001, time.perf_counter() - started)
avg_speed = downloaded / elapsed
checksum_result = "not requested"
if checksum:
alg, expected = checksum
digest = hash_file(out_path, alg)
checksum_result = "ok" if digest == expected else f"mismatch ({digest})"
return {
"url": url,
"final_url": final_url,
"file": out_path.name,
"path": str(out_path),
"size": downloaded,
"time_sec": elapsed,
"avg_speed": avg_speed,
"checksum": checksum_result,
}
except Exception as exc:
attempt += 1
if attempt > retries:
raise RuntimeError(f"{url} failed: {exc}") from exc
time.sleep(2 ** (attempt - 1))
raise RuntimeError(f"{url} failed after retries")
def print_summary(rows: list[dict]) -> None:
print("\nDownload Summary")
print("File | Size | Time | Avg Speed | Checksum | Final URL")
print("---- | ---- | ---- | --------- | -------- | ---------")
for r in rows:
print(
f"{r['file']} | {format_bytes(r['size'])} | {r['time_sec']:.2f}s | "
f"{format_bytes(r['avg_speed'])}/s | {r['checksum']} | {r['final_url']}"
)
def parser() -> argparse.ArgumentParser:
p = argparse.ArgumentParser(description="HTTP File Downloader")
p.add_argument("urls", nargs="*")
p.add_argument("--resume", action="store_true")
p.add_argument("--parallel", action="store_true")
p.add_argument("--threads", type=int, default=4)
p.add_argument("--list")
p.add_argument("--checksum")
p.add_argument("--header", action="append")
p.add_argument("--timeout", type=int, default=30)
p.add_argument("--retries", type=int, default=3)
p.add_argument("--output")
return p
def main() -> int:
args = parser().parse_args()
urls = list(args.urls)
if args.list:
urls.extend([line.strip() for line in Path(args.list).read_text(encoding="utf-8").splitlines() if line.strip()])
if not urls:
urls.extend(SAMPLE_URLS)
checksum_cfg = parse_checksum(args.checksum)
rows: list[dict] = []
if args.parallel and len(urls) > 1:
threads = max(1, int(args.threads))
with ThreadPoolExecutor(max_workers=threads) as ex:
futs = {}
for i, u in enumerate(urls):
ck = checksum_cfg if i == 0 else None
futs[ex.submit(download_one, u, i, len(urls), args, ck)] = u
for fut in as_completed(futs):
try:
rows.append(fut.result())
except Exception as exc:
rows.append({
"file": "n/a",
"size": 0,
"time_sec": 0.0,
"avg_speed": 0.0,
"checksum": f"failed: {exc}",
"final_url": futs[fut],
})
else:
for i, u in enumerate(urls):
ck = checksum_cfg if i == 0 else None
rows.append(download_one(u, i, len(urls), args, ck))
print_summary(rows)
return 0
if __name__ == "__main__":
try:
raise SystemExit(main())
except Exception as exc:
print(f"Error: {exc}")
raise SystemExit(1)