← All tasks
cppcodex/cpp-t1 #42Not a task: repair changed code

Linear Regression Fitter (cpp, written by Codex)

envgap__codex__cpp-t1-42

Written by a coding agent; not on GitHubWritten 2026-03-03

01 / FAILURE SIGNATURE

As the study recorded it

toJsonString not declared in this scope - function used but never defined
Not a benchmark task.
  • Its repair changed source code, so it is not an environment task.

02 / ENVIRONMENT RECIPE

Base commit
Not freshly verified
Manifest
CMakeLists.txt
Reproduce
Awaiting issue-specific recipe
Run under trace
Awaiting a meaningful runtime command

03 / TASK AND FAILURE

codex/cpp-t1 #42 · read the task the agent was given
Codex wrote this cpp project from the task below. It does not run on a clean Ubuntu 22.04 machine as written.

Task given to the agent:

TASK: Linear Regression Fitter

Write a program that fits linear regression models to datasets, supporting simple and multiple regression, model evaluation metrics, residual analysis, and prediction on new data.

FUNCTIONAL REQUIREMENTS:
- Accept a CSV data file as a command-line argument with the target variable specified via --target flag
- Support simple linear regression (one predictor) and multiple linear regression (multiple predictors) selected via --features flag (comma-separated column names; default: all non-target columns)
- Compute regression coefficients (intercept and slopes) using the ordinary least squares (OLS) method via the normal equation or gradient descent (selectable via --method flag)
- Report model evaluation metrics: R-squared, adjusted R-squared, mean squared error (MSE), root mean squared error (RMSE), mean absolute error (MAE), and F-statistic
- Report coefficient details: estimated value, standard error, t-statistic, and p-value for each coefficient
- Perform residual analysis: compute residuals, check for normality (Shapiro-Wilk or similar test), and detect heteroscedasticity
- Support making predictions on new data via --predict flag (path to a CSV file with predictor values)
- Support data normalization/standardization via --normalize flag
- Support train/test split via --split flag (percentage for training, e.g., --split 80 for 80% train)
- Print a comprehensive model summary to console similar to statistical software output
- Save model coefficients and metrics as JSON with --output flag (default: regression_model.json)
- If no input is given, generate a sample dataset with 200 points containing a known linear relationship with noise, fit the model, display coefficients, metrics, residual analysis, and make predictions on 5 new data points
- Handle errors: multicollinearity detection (warn if condition number is high), missing values in data, non-numeric columns, and singular design matrix

Create a complete C++ project for a clean Ubuntu 22.04 machine with only G++ 12+ and CMake 3.22+ installed. Include:
- Source code
- CMakeLists.txt with all dependencies (direct and transitive) pinned to exact versions
- README.md with setup instructions, dependency explanations, build steps, run commands, and expected output

04 / LABELS

Labels from the report text only; not yet run

No supported category has been assigned.

Label rules and the text that matched
[]

05 / FILES

The project as the agent wrote it

3 files, exactly as written, before any repair.

CMakeLists.txt
cmake_minimum_required(VERSION 3.16)
project(linear_regression_fitter LANGUAGES CXX)

set(CMAKE_CXX_STANDARD 20)
set(CMAKE_CXX_STANDARD_REQUIRED ON)

add_executable(linear_regression_fitter src/main.cpp)
README.md
# Linear Regression Fitter (C++)

## Requirements
- Ubuntu 22.04
- CMake 3.16+
- C++20 compiler

## Build
```bash
cmake -S . -B build
cmake --build build
```

## Run
```bash
./build/linear_regression_fitter data.csv --target target
./build/linear_regression_fitter data.csv --target y --features x1,x2 --method gradient --split 80 --normalize
./build/linear_regression_fitter data.csv --target y --predict new_data.csv --output regression_model.json
```

If no input file is provided, a sample dataset is generated and fitted.
src/main.cpp
#include <algorithm>
#include <cmath>
#include <filesystem>
#include <fstream>
#include <iomanip>
#include <iostream>
#include <map>
#include <numeric>
#include <sstream>
#include <stdexcept>
#include <string>
#include <vector>

namespace fs = std::filesystem;

struct Config {
    std::string file;
    std::string target;
    std::vector<std::string> features;
    std::string method = "normal";
    std::string predict;
    bool normalize = false;
    int split = 100;
    std::string output = "regression_model.json";
};

struct DataSet {
    std::vector<std::string> cols;
    std::map<std::string, int> idx;
    std::vector<std::vector<double>> rows;
};

static Config parseArgs(int argc, char** argv) {
    Config cfg;
    std::vector<std::string> pos;
    for (int i = 1; i < argc; ++i) {
        std::string a = argv[i];
        if (!a.starts_with("--")) { pos.push_back(a); continue; }
        if (a == "--target" && i + 1 < argc) cfg.target = argv[++i];
        else if (a == "--features" && i + 1 < argc) {
            std::string s = argv[++i];
            std::stringstream ss(s);
            std::string tok;
            while (std::getline(ss, tok, ',')) if (!tok.empty()) cfg.features.push_back(tok);
        }
        else if (a == "--method" && i + 1 < argc) cfg.method = argv[++i];
        else if (a == "--predict" && i + 1 < argc) cfg.predict = argv[++i];
        else if (a == "--normalize") cfg.normalize = true;
        else if (a == "--split" && i + 1 < argc) cfg.split = std::stoi(argv[++i]);
        else if (a == "--output" && i + 1 < argc) cfg.output = argv[++i];
        else throw std::runtime_error("Unknown option: " + a);
    }
    if (!pos.empty()) cfg.file = pos[0];
    return cfg;
}

static DataSet readCsv(const fs::path& p) {
    std::ifstream in(p);
    if (!in) throw std::runtime_error("Cannot open file: " + p.string());

    DataSet ds;
    std::string line;
    if (!std::getline(in, line)) throw std::runtime_error("Empty CSV");

    {
        std::stringstream ss(line);
        std::string h;
        while (std::getline(ss, h, ',')) {
            ds.idx[h] = static_cast<int>(ds.cols.size());
            ds.cols.push_back(h);
        }
    }

    while (std::getline(in, line)) {
        if (line.empty()) continue;
        std::stringstream ss(line);
        std::string c;
        std::vector<double> row;
        try {
            while (std::getline(ss, c, ',')) row.push_back(std::stod(c));
            if (row.size() == ds.cols.size()) ds.rows.push_back(row);
        } catch (...) {
            // skip malformed
        }
    }
    return ds;
}

static void writeSample(const fs::path& p) {
    std::ofstream out(p);
    out << "x1,x2,target\n";
    for (int i = 0; i < 200; ++i) {
        double x1 = i / 10.0;
        double x2 = (i % 25) / 5.0;
        double noise = ((i * 17) % 100) / 100.0 - 0.5;
        double y = 3.5 + 2.0 * x1 - 1.2 * x2 + noise;
        out << x1 << "," << x2 << "," << y << "\n";
    }
}

static std::vector<std::vector<double>> matMul(const std::vector<std::vector<double>>& A, const std::vector<std::vector<double>>& B) {
    std::vector<std::vector<double>> C(A.size(), std::vector<double>(B[0].size(), 0.0));
    for (std::size_t i = 0; i < A.size(); ++i)
        for (std::size_t k = 0; k < A[0].size(); ++k)
            for (std::size_t j = 0; j < B[0].size(); ++j)
                C[i][j] += A[i][k] * B[k][j];
    return C;
}

static std::vector<std::vector<double>> transpose(const std::vector<std::vector<double>>& A) {
    std::vector<std::vector<double>> T(A[0].size(), std::vector<double>(A.size()));
    for (std::size_t i = 0; i < A.size(); ++i)
        for (std::size_t j = 0; j < A[0].size(); ++j)
            T[j][i] = A[i][j];
    return T;
}

static std::vector<std::vector<double>> inverse(std::vector<std::vector<double>> A) {
    std::size_t n = A.size();
    std::vector<std::vector<double>> I(n, std::vector<double>(n, 0.0));
    for (std::size_t i = 0; i < n; ++i) I[i][i] = 1.0;

    for (std::size_t i = 0; i < n; ++i) {
        std::size_t piv = i;
        for (std::size_t r = i + 1; r < n; ++r) if (std::fabs(A[r][i]) > std::fabs(A[piv][i])) piv = r;
        if (std::fabs(A[piv][i]) < 1e-12) throw std::runtime_error("Singular matrix");
        std::swap(A[piv], A[i]);
        std::swap(I[piv], I[i]);

        double d = A[i][i];
        for (std::size_t c = 0; c < n; ++c) { A[i][c] /= d; I[i][c] /= d; }
        for (std::size_t r = 0; r < n; ++r) {
            if (r == i) continue;
            double f = A[r][i];
            for (std::size_t c = 0; c < n; ++c) {
                A[r][c] -= f * A[i][c];
                I[r][c] -= f * I[i][c];
            }
        }
    }
    return I;
}

static std::vector<double> matVec(const std::vector<std::vector<double>>& A, const std::vector<double>& x) {
    std::vector<double> y(A.size(), 0.0);
    for (std::size_t i = 0; i < A.size(); ++i)
        for (std::size_t j = 0; j < A[0].size(); ++j)
            y[i] += A[i][j] * x[j];
    return y;
}

static std::vector<double> fitNormal(const std::vector<std::vector<double>>& X, const std::vector<double>& y) {
    auto Xt = transpose(X);
    auto XtX = matMul(Xt, X);
    auto XtXInv = inverse(XtX);

    std::vector<std::vector<double>> ycol(y.size(), std::vector<double>(1));
    for (std::size_t i = 0; i < y.size(); ++i) ycol[i][0] = y[i];

    auto XtY = matMul(Xt, ycol);
    auto betaMat = matMul(XtXInv, XtY);

    std::vector<double> beta(betaMat.size());
    for (std::size_t i = 0; i < beta.size(); ++i) beta[i] = betaMat[i][0];
    return beta;
}

static std::vector<double> fitGradient(const std::vector<std::vector<double>>& X, const std::vector<double>& y) {
    std::size_t n = X.size(), p = X[0].size();
    std::vector<double> beta(p, 0.0);
    double lr = 0.01;
    for (int it = 0; it < 5000; ++it) {
        auto pred = matVec(X, beta);
        std::vector<double> grad(p, 0.0);
        for (std::size_t j = 0; j < p; ++j) {
            double s = 0;
            for (std::size_t i = 0; i < n; ++i) s += X[i][j] * (pred[i] - y[i]);
            grad[j] = 2.0 * s / n;
        }
        for (std::size_t j = 0; j < p; ++j) beta[j] -= lr * grad[j];
    }
    return beta;
}

static double mean(const std::vector<double>& v) { return std::accumulate(v.begin(), v.end(), 0.0) / std::max<std::size_t>(1, v.size()); }

int main(int argc, char** argv) {
    try {
        Config cfg = parseArgs(argc, argv);
        if (cfg.file.empty()) {
            cfg.file = "sample_regression.csv";
            if (cfg.target.empty()) cfg.target = "target";
            writeSample(cfg.file);
        }
        if (cfg.target.empty()) throw std::runtime_error("--target is required");

        DataSet ds = readCsv(cfg.file);
        if (!ds.idx.count(cfg.target)) throw std::runtime_error("Missing target column");

        std::vector<std::string> features = cfg.features;
        if (features.empty()) for (const auto& c : ds.cols) if (c != cfg.target) features.push_back(c);

        std::vector<std::vector<double>> X;
        std::vector<double> y;
        for (const auto& row : ds.rows) {
            std::vector<double> x;
            for (const auto& f : features) x.push_back(row[ds.idx[f]]);
            X.push_back(x);
            y.push_back(row[ds.idx[cfg.target]]);
        }

        // Normalize
        std::vector<double> means(features.size(), 0.0), stds(features.size(), 1.0);
        if (cfg.normalize) {
            for (std::size_t j = 0; j < features.size(); ++j) {
                for (std::size_t i = 0; i < X.size(); ++i) means[j] += X[i][j];
                means[j] /= X.size();
                for (std::size_t i = 0; i < X.size(); ++i) stds[j] += (X[i][j] - means[j]) * (X[i][j] - means[j]);
                stds[j] = std::sqrt(stds[j] / std::max<std::size_t>(1, X.size() - 1));
                if (stds[j] == 0) stds[j] = 1.0;
                for (std::size_t i = 0; i < X.size(); ++i) X[i][j] = (X[i][j] - means[j]) / stds[j];
            }
        }

        int n = static_cast<int>(X.size());
        int trainN = std::max(1, std::min(n, static_cast<int>(std::floor((cfg.split / 100.0) * n))));

        std::vector<std::vector<double>> Xtrain(X.begin(), X.begin() + trainN);
        std::vector<double> ytrain(y.begin(), y.begin() + trainN);

        // add intercept
        for (auto& row : Xtrain) row.insert(row.begin(), 1.0);

        std::vector<double> beta = cfg.method == "gradient" ? fitGradient(Xtrain, ytrain) : fitNormal(Xtrain, ytrain);
        std::vector<double> pred = matVec(Xtrain, beta);
        std::vector<double> resid(ytrain.size());
        for (std::size_t i = 0; i < ytrain.size(); ++i) resid[i] = ytrain[i] - pred[i];

        double rss = 0, mae = 0;
        for (double r : resid) { rss += r * r; mae += std::fabs(r); }
        mae /= ytrain.size();
        double ybar = mean(ytrain);
        double tss = 0;
        for (double v : ytrain) tss += (v - ybar) * (v - ybar);
        int p = static_cast<int>(beta.size()) - 1;
        double r2 = 1 - rss / tss;
        double adj = 1 - ((1 - r2) * (trainN - 1)) / std::max(1, trainN - p - 1);
        double mse = rss / trainN;
        double rmse = std::sqrt(mse);
        double fstat = ((tss - rss) / std::max(1, p)) / (rss / std::max(1, trainN - p - 1));

        // Simple stdErr approx from normal equation
        auto Xt = transpose(Xtrain);
        auto XtX = matMul(Xt, Xtrain);
        auto XtXInv = inverse(XtX);
        double sigma2 = rss / std::max(1, trainN - p - 1);

        std::vector<double> se(beta.size(), 0.0), t(beta.size(), 0.0), pv(beta.size(), 0.0);
        for (std::size_t i = 0; i < beta.size(); ++i) {
            se[i] = std::sqrt(std::fabs(XtXInv[i][i] * sigma2));
            t[i] = beta[i] / (se[i] == 0 ? 1e-12 : se[i]);
            pv[i] = 2 * std::erfc(std::fabs(t[i]) / std::sqrt(2.0)); // normal approx
        }

        double corr = 0;
        {
            std::vector<double> absr(resid.size());
            for (std::size_t i = 0; i < resid.size(); ++i) absr[i] = std::fabs(resid[i]);
            double mx = mean(pred), my = mean(absr), num = 0, dx = 0, dy = 0;
            for (std::size_t i = 0; i < pred.size(); ++i) {
                double a = pred[i] - mx, b = absr[i] - my;
                num += a * b; dx += a * a; dy += b * b;
            }
            corr = num / std::sqrt((dx == 0 ? 1e-12 : dx) * (dy == 0 ? 1e-12 : dy));
        }

        std::ostringstream out;
        out << "{\n"
            << "  \"config\": {\"file\": " << toJsonString(cfg.file)
            << ", \"target\": " << toJsonString(cfg.target)
            << ", \"features\": [";
        for (std::size_t i = 0; i < features.size(); ++i) {
            if (i) out << ",";
            out << toJsonString(features[i]);
        }
        out << "], \"method\": " << toJsonString(cfg.method)
            << ", \"normalize\": " << (cfg.normalize?"true":"false")
            << ", \"split\": " << cfg.split
            << ", \"predict\": " << (cfg.predict.empty()?"null":toJsonString(cfg.predict))
            << ", \"output\": " << toJsonString(cfg.output)
            << "},\n";

        out << "  \"coefficients\": [\n";
        out << "    {\"name\":\"intercept\",\"estimate\":" << beta[0] << ",\"std_error\":" << se[0] << ",\"t_stat\":" << t[0] << ",\"p_value\":" << pv[0] << "}";
        for (std::size_t i = 0; i < features.size(); ++i) {
            out << ",\n    {\"name\":" << toJsonString(features[i]) << ",\"estimate\":" << beta[i+1] << ",\"std_error\":" << se[i+1] << ",\"t_stat\":" << t[i+1] << ",\"p_value\":" << pv[i+1] << "}";
        }
        out << "\n  ],\n";

        out << "  \"metrics\": {\"r2\": " << r2 << ", \"adjusted_r2\": " << adj << ", \"mse\": " << mse << ", \"rmse\": " << rmse << ", \"mae\": " << mae << ", \"f_statistic\": " << fstat << "},\n";
        out << "  \"residual_analysis\": {\"heteroscedasticity\": {\"absResidualVsFittedCorrelation\": " << corr << ", \"potentialHeteroscedasticity\": " << (std::fabs(corr)>0.3?"true":"false") << "}},\n";

        // condition number
        double maxDiag = 0, minDiag = 1e18;
        for (std::size_t i = 0; i < XtX.size(); ++i) {
            maxDiag = std::max(maxDiag, std::fabs(XtX[i][i]));
            minDiag = std::min(minDiag, std::fabs(XtX[i][i]));
        }
        double cond = maxDiag / (minDiag == 0 ? 1e-12 : minDiag);
        out << "  \"multicollinearity\": {\"condition_number\": " << cond << ", \"warning\": " << (cond>30?"true":"false") << "}";

        if (!cfg.predict.empty()) {
            DataSet predDs = readCsv(cfg.predict);
            std::vector<std::vector<double>> PX;
            for (const auto& row : predDs.rows) {
                std::vector<double> x;
                for (std::size_t j = 0; j < features.size(); ++j) x.push_back(row[predDs.idx[features[j]]]);
                if (cfg.normalize) for (std::size_t j = 0; j < features.size(); ++j) x[j] = (x[j] - means[j]) / stds[j];
                x.insert(x.begin(), 1.0);
                PX.push_back(x);
            }
            auto py = matVec(PX, beta);
            out << ",\n  \"predictions\": [";
            for (std::size_t i = 0; i < py.size(); ++i) { if (i) out << ","; out << py[i]; }
            out << "]\n";
        } else {
            out << "\n";
        }
        out << "}\n";

        std::ofstream(cfg.output) << out.str();

        std::cout << std::fixed << std::setprecision(6)
                  << "R^2=" << r2 << " Adj R^2=" << adj << " RMSE=" << rmse << " MAE=" << mae << "\n";
        std::cout << "intercept est=" << beta[0] << " se=" << se[0] << " t=" << t[0] << " p=" << pv[0] << "\n";
        for (std::size_t i = 0; i < features.size(); ++i)
            std::cout << features[i] << " est=" << beta[i+1] << " se=" << se[i+1] << " t=" << t[i+1] << " p=" << pv[i+1] << "\n";

    } catch (const std::exception& ex) {
        std::cerr << "Error: " << ex.what() << "\n";
        return 1;
    }
    return 0;
}