Multi-Format Data Converter (python, written by Codex)
envgap__codex__python-t1-8
Written by a coding agent; not on GitHubWritten 2026-03-02
01 / FAILURE SIGNATURE
As the study recorded it
None
Not a benchmark task.
- The project already builds and runs before the fix, so there is nothing to repair.
02 / ENVIRONMENT RECIPE
- Base commit
Not freshly verified- Manifest
requirements.txt- Reproduce
Awaiting issue-specific recipe- Run under trace
Awaiting a meaningful runtime command
03 / TASK AND FAILURE
codex/python-t1 #8 · read the task the agent was given
Codex wrote this python project from the task below. It installed and ran on a clean Ubuntu 22.04 machine as written.
Task given to the agent:
TASK: Multi-Format Data Converter
Write a program that converts data files between CSV, JSON, XML, and YAML formats while preserving data types, nested structures, and handling large files efficiently.
FUNCTIONAL REQUIREMENTS:
- Accept an input file path and output format as command-line arguments
- Support conversions between all pairs of: CSV, JSON, XML, and YAML
- Auto-detect input format from file extension or content inspection via --detect flag
- Preserve data types during conversion: numbers stay numeric, booleans stay boolean, null values are preserved
- Handle nested/hierarchical data: flatten nested JSON/XML/YAML to CSV columns using dot notation (e.g., address.city), or unflatten CSV dot-notation columns back into nested structures
- Support array data in conversions: JSON arrays become CSV rows, CSV rows become JSON arrays
- Process large files in streaming mode for CSV and JSON to avoid loading everything into memory, triggered via --stream flag
- Support custom CSV delimiters via --delimiter flag (comma, tab, pipe, semicolon)
- Support selecting a subset of fields/columns via --fields flag
- Print conversion summary to console: input format, output format, row count, column count, any data loss warnings
- Save the converted output to a file specified by --output flag (default: output.{format})
- If no input file is given, generate a sample dataset with nested objects, arrays, mixed types, and null values in JSON format, then convert it to all other formats
- Handle encoding differences (UTF-8, Latin-1) and BOM markers gracefully
Create a complete Python project for a clean Ubuntu 22.04 machine with only Python 3.10+ installed. Include:
- Source code
- requirements.txt with all dependencies (direct and transitive) pinned to exact versions
- README.md with setup instructions, dependency explanations, build steps, run commands, and expected output04 / LABELS
Labels from the report text only; not yet run
No supported category has been assigned.
Label rules and the text that matched
[]
05 / FILES
The project as the agent wrote it
3 files, exactly as written, before any repair.
README.md
# Multi-Format Data Converter (Python) Converts data between `CSV`, `JSON`, `XML`, and `YAML` while preserving types, nested structures (dot notation for CSV), and optional stream-mode parsing. ## Requirements - Ubuntu 22.04 - Python 3.10+ ## Dependencies - Direct: - `PyYAML==6.0.2` - `xmltodict==0.13.0` - Transitive: - none All pinned in `requirements.txt`. ## Setup ```bash python -m pip install -r requirements.txt ``` ## Run General form: ```bash python src/main.py <input-file> <output-format> [--output file] [--detect] [--stream] [--delimiter comma|tab|pipe|semicolon] [--fields a,b,c] ``` Examples: ```bash python src/main.py ./data.json csv --output out.csv --fields id,name,address.city python src/main.py ./data.csv json --delimiter semicolon --detect python src/main.py ./data.ndjson yaml --stream ``` No input (generates sample JSON and converts to all other formats): ```bash python src/main.py ``` ## Notes - Stream mode supports CSV and NDJSON-style JSON input. - CSV uses dot notation for nested fields during flattening. - Handles UTF-8 BOM and Latin-1 fallback decoding.
requirements.txt
PyYAML==6.0.2 xmltodict==0.13.0
src/main.py
#!/usr/bin/env python3
import argparse
import csv
import json
from pathlib import Path
from typing import Any
import xmltodict
import yaml
FORMAT_SET = {"csv", "json", "xml", "yaml"}
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description="Multi-Format Data Converter")
parser.add_argument("input_file", nargs="?", help="Input file path")
parser.add_argument("output_format", nargs="?", help="Target format: csv|json|xml|yaml")
parser.add_argument("--format", dest="format_opt", help="Target format if not provided as positional")
parser.add_argument("--detect", action="store_true", help="Auto-detect input format from extension/content")
parser.add_argument("--stream", action="store_true", help="Stream mode for CSV and NDJSON JSON input")
parser.add_argument("--delimiter", default="comma", help="CSV delimiter: comma|tab|pipe|semicolon or literal")
parser.add_argument("--fields", help="Comma-separated dot-notation field subset")
parser.add_argument("--output", help="Output file path")
return parser.parse_args()
def normalize_delimiter(value: str) -> str:
mapping = {
",": ",",
"comma": ",",
"\\t": "\t",
"tab": "\t",
"|": "|",
"pipe": "|",
";": ";",
"semicolon": ";",
}
return mapping.get(value, value[0] if value else ",")
def read_text_with_encoding(path: Path) -> tuple[str, str]:
raw = path.read_bytes()
try:
text = raw.decode("utf-8-sig")
return text, "utf-8"
except UnicodeDecodeError:
return raw.decode("latin-1"), "latin-1"
def detect_format(path: Path, text: str, detect_flag: bool) -> str:
if not detect_flag:
ext = path.suffix.lower()
if ext == ".csv":
return "csv"
if ext in {".json", ".jsonl", ".ndjson"}:
return "json"
if ext == ".xml":
return "xml"
if ext in {".yaml", ".yml"}:
return "yaml"
trimmed = text.strip()
if trimmed.startswith("{") or trimmed.startswith("["):
return "json"
if trimmed.startswith("<"):
return "xml"
if ":" in trimmed.splitlines()[0] if trimmed else False:
return "yaml"
return "csv"
def parse_primitive(value: Any) -> Any:
if value is None:
return None
s = str(value).strip()
if s == "":
return None
if s.lower() == "null":
return None
if s.lower() == "true":
return True
if s.lower() == "false":
return False
if s.lstrip("+-").isdigit():
try:
return int(s)
except ValueError:
pass
try:
if "." in s:
return float(s)
except ValueError:
pass
return value
def flatten_obj(value: Any, prefix: str = "", out: dict[str, Any] | None = None) -> dict[str, Any]:
if out is None:
out = {}
if value is None:
if prefix:
out[prefix] = None
return out
if isinstance(value, list):
if not value and prefix:
out[prefix] = []
return out
for idx, item in enumerate(value):
key = f"{prefix}.{idx}" if prefix else str(idx)
flatten_obj(item, key, out)
return out
if isinstance(value, dict):
if not value and prefix:
out[prefix] = {}
return out
for k, v in value.items():
key = f"{prefix}.{k}" if prefix else str(k)
flatten_obj(v, key, out)
return out
if prefix:
out[prefix] = value
return out
def assign_path(obj: dict[str, Any], parts: list[str], value: Any) -> None:
current: Any = obj
for idx, part in enumerate(parts):
is_last = idx == len(parts) - 1
is_index = part.isdigit()
if is_last:
if isinstance(current, list) and is_index:
i = int(part)
while len(current) <= i:
current.append(None)
current[i] = value
else:
current[part] = value
return
next_is_index = parts[idx + 1].isdigit()
if isinstance(current, list) and is_index:
i = int(part)
while len(current) <= i:
current.append([] if next_is_index else {})
if current[i] is None:
current[i] = [] if next_is_index else {}
current = current[i]
else:
if part not in current or current[part] is None:
current[part] = [] if next_is_index else {}
current = current[part]
def unflatten_row(flat: dict[str, Any]) -> dict[str, Any]:
out: dict[str, Any] = {}
for k, v in flat.items():
if "." not in k:
out[k] = v
else:
assign_path(out, k.split("."), v)
return out
def normalize_rows(data: Any) -> list[dict[str, Any]]:
if isinstance(data, list):
return [item if isinstance(item, dict) else {"value": item} for item in data]
if isinstance(data, dict):
return [data]
return [{"value": data}]
def apply_fields(rows: list[dict[str, Any]], fields: list[str]) -> list[dict[str, Any]]:
if not fields:
return rows
out = []
for row in rows:
flat = flatten_obj(row)
selected = {f: flat[f] for f in fields if f in flat}
out.append(unflatten_row(selected))
return out
def parse_csv_text(text: str, delimiter: str) -> list[dict[str, Any]]:
lines = [line for line in text.replace("\r\n", "\n").replace("\r", "\n").split("\n") if line.strip()]
if not lines:
return []
reader = csv.DictReader(lines, delimiter=delimiter)
rows = []
for row in reader:
parsed = {k: parse_primitive(v) for k, v in row.items() if k is not None}
rows.append(unflatten_row(parsed))
return rows
def parse_csv_stream(path: Path, delimiter: str) -> list[dict[str, Any]]:
rows = []
with path.open("r", encoding="utf-8-sig", newline="") as f:
reader = csv.DictReader(f, delimiter=delimiter)
for row in reader:
parsed = {k: parse_primitive(v) for k, v in row.items() if k is not None}
rows.append(unflatten_row(parsed))
return rows
def parse_json_stream(path: Path) -> list[dict[str, Any]]:
rows = []
with path.open("r", encoding="utf-8-sig") as f:
for line in f:
trimmed = line.strip()
if not trimmed:
continue
obj = json.loads(trimmed)
if isinstance(obj, dict):
rows.append(obj)
else:
rows.append({"value": obj})
return rows
def parse_input(path: Path, input_format: str, delimiter: str, stream_mode: bool, warnings: list[str]) -> list[dict[str, Any]]:
if stream_mode and input_format == "csv":
return parse_csv_stream(path, delimiter)
if stream_mode and input_format == "json":
try:
return parse_json_stream(path)
except Exception:
warnings.append("JSON stream mode expects NDJSON; falling back to full parse.")
text, _ = read_text_with_encoding(path)
if input_format == "csv":
return parse_csv_text(text, delimiter)
if input_format == "json":
return normalize_rows(json.loads(text))
if input_format == "yaml":
return normalize_rows(yaml.safe_load(text))
if input_format == "xml":
parsed = xmltodict.parse(text)
if "root" in parsed and isinstance(parsed["root"], dict) and "item" in parsed["root"]:
return normalize_rows(parsed["root"]["item"])
return normalize_rows(parsed)
raise ValueError(f"Unsupported input format: {input_format}")
def serialize_csv(rows: list[dict[str, Any]], delimiter: str) -> str:
flat_rows = [flatten_obj(row) for row in rows]
headers = sorted({k for row in flat_rows for k in row.keys()})
output = []
output.append(delimiter.join(headers))
for row in flat_rows:
values = []
for h in headers:
v = row.get(h)
text = "" if v is None else str(v)
if delimiter in text or '"' in text or "\n" in text:
text = '"' + text.replace('"', '""') + '"'
values.append(text)
output.append(delimiter.join(values))
return "\n".join(output) + "\n"
def serialize_json(rows: list[dict[str, Any]]) -> str:
return json.dumps(rows, indent=2) + "\n"
def serialize_yaml(rows: list[dict[str, Any]]) -> str:
return yaml.safe_dump(rows, sort_keys=False)
def serialize_xml(rows: list[dict[str, Any]]) -> str:
payload = {"root": {"item": rows}}
return xmltodict.unparse(payload, pretty=True)
def write_output(path: Path, output_format: str, rows: list[dict[str, Any]], delimiter: str) -> None:
if output_format == "csv":
content = serialize_csv(rows, delimiter)
elif output_format == "json":
content = serialize_json(rows)
elif output_format == "yaml":
content = serialize_yaml(rows)
elif output_format == "xml":
content = serialize_xml(rows)
else:
raise ValueError(f"Unsupported output format: {output_format}")
path.write_text(content, encoding="utf-8")
def count_columns(rows: list[dict[str, Any]]) -> int:
columns = set()
for row in rows:
columns.update(flatten_obj(row).keys())
return len(columns)
def print_summary(input_format: str, output_format: str, rows: list[dict[str, Any]], warnings: list[str]) -> None:
print("Conversion Summary")
print("==================")
print(f"Input format : {input_format}")
print(f"Output format: {output_format}")
print(f"Row count : {len(rows)}")
print(f"Column count : {count_columns(rows)}")
print(f"Warnings : {' | '.join(warnings) if warnings else 'none'}")
def sample_data() -> list[dict[str, Any]]:
return [
{
"id": 1,
"name": "Alice",
"active": True,
"score": 97.5,
"address": {"city": "Austin", "zip": "73301"},
"tags": ["premium", "beta"],
"orders": [{"id": "o1", "amount": 39.95}, {"id": "o2", "amount": 12}],
"last_login": None,
},
{
"id": 2,
"name": "Bob",
"active": False,
"score": 88,
"address": {"city": "Berlin", "zip": "10115"},
"tags": ["standard"],
"orders": [{"id": "o3", "amount": 120.1}],
"last_login": "2026-01-01T10:00:00Z",
},
]
def run_single(input_path: Path, output_format: str, args: argparse.Namespace) -> int:
delimiter = normalize_delimiter(args.delimiter)
warnings: list[str] = []
text, encoding = read_text_with_encoding(input_path)
input_format = detect_format(input_path, text, args.detect)
if input_format not in FORMAT_SET or output_format not in FORMAT_SET:
raise ValueError("Supported formats: csv|json|xml|yaml")
if encoding != "utf-8":
warnings.append(f"Input decoded as {encoding}")
rows = parse_input(input_path, input_format, delimiter, args.stream, warnings)
fields = [f.strip() for f in args.fields.split(",")] if args.fields else []
rows = apply_fields(rows, [f for f in fields if f])
if output_format == "csv":
warnings.append("Nested objects/arrays are flattened to dot-notation columns.")
output_path = Path(args.output).resolve() if args.output else Path(f"output.{output_format}").resolve()
write_output(output_path, output_format, rows, delimiter)
print_summary(input_format, output_format, rows, warnings)
print(f"Output file : {output_path}")
return 0
def main() -> int:
args = parse_args()
if not args.input_file:
data = sample_data()
sample_path = Path("sample_data.json").resolve()
sample_path.write_text(json.dumps(data, indent=2) + "\n", encoding="utf-8")
print(f"No input provided. Generated sample dataset: {sample_path}")
for fmt in ("csv", "xml", "yaml"):
output_path = Path(f"output.{fmt}").resolve()
write_output(output_path, fmt, data, normalize_delimiter(args.delimiter))
print_summary("json", fmt, data, [])
print(f"Output file : {output_path}")
print("")
return 0
input_path = Path(args.input_file).resolve()
if not input_path.exists():
print(f"Input file not found: {input_path}")
return 1
output_format = (args.output_format or args.format_opt or "").lower()
if output_format not in FORMAT_SET:
print("Provide output format as positional or --format (csv|json|xml|yaml).")
return 1
try:
return run_single(input_path, output_format, args)
except Exception as exc:
print(f"Conversion failed: {exc}")
return 1
if __name__ == "__main__":
raise SystemExit(main())