← All tasks
javaclaude-code/java-t1 #5Lite task

Log File Pattern Analyzer (java, written by Claude Code)

envgap__claude-code__java-t1-5

Written by a coding agent; not on GitHubWritten 2026-02-27

01 / FAILURE SIGNATURE

Captured in a clean container

error: classes the program uses are missing from the class path it runs with

02 / ENVIRONMENT RECIPE

Base commit
8c6eda42c1b59ffb6e98a6ccfef9b4e5df6e71cc
Manifest
pom.xml
Reproduce
jar=$(ls target/*-jar-with-dependencies.jar target/*-shaded.jar target/*-all.jar 2>/dev/null | head -n1); [ -n "$jar" ] || jar=$(ls -S target/*.jar 2>/dev/null | grep -v -e '/original-' -e '-sources.jar$' -e '-javadoc.jar$' -e '-tests.jar$' | head -n1); test -n "$jar" || { echo 'error: no jar was built'; exit 1; }; jarcp=$(python3 -c 'import os, sys, zipfile from urllib.parse import unquote jar = sys.argv[1] try: text = zipfile.ZipFile(jar).read("META-INF/MANIFEST.MF").decode("utf-8", "replace") except (KeyError, OSError, zipfile.BadZipFile): text = "" text = text.replace("\r\n", "\n").replace("\r", "\n").replace("\n ", "") found = [line.split(":", 1)[1].split() for line in text.split("\n") if line.lower().startswith("class-path:")] entries = [os.path.join(os.path.dirname(jar), unquote(entry)) for entry in (found[0] if found else [])] print(":".join([jar] + [entry for entry in entries if os.path.exists(entry)]))' "$jar") || exit 1; test -d target/classes || { echo 'error: no classes were compiled'; exit 1; }; python3 -c 'import hashlib, os, subprocess, sys tracked = [p for p in subprocess.run(["git", "ls-files", "-z", "--", "*.java"], capture_output=True).stdout.decode().split("\0") if p] digest = lambda p: hashlib.sha256(open(p, "rb").read()).hexdigest() own = {digest(p) for p in tracked if os.path.isfile(p)} names = {os.path.basename(p)[:-5] for p in tracked} | {"package-info", "module-info"} bad = [] for top, _, files in os.walk("target"): for name in files: path = os.path.join(top, name) if name.endswith(".java") and digest(path) not in own: bad.append(path) elif top.startswith(os.path.join("target", "classes")) and name.endswith(".class") and name[:-6].split("$")[0] not in names: bad.append(path) if bad: print("\n".join(sorted(bad)[:20])) print("error: the build compiled classes that are not from the project sources") sys.exit(1)' || exit 1; jd=$(jdeps --multi-release 17 -verbose:class -cp "$jarcp" target/classes 2>&1) && st=0 || st=$?; missing=$(printf '%s\n' "$jd" | grep 'not found' || true); if [ $st -ne 0 ]; then printf '%s\n' "$jd" | tail -n 20; echo 'error: jdeps could not read the classes'; exit 1; fi; if [ -n "$missing" ]; then printf '%s\n' "$missing"; echo 'error: classes the program uses are missing from the class path it runs with'; exit 1; fi
Run under trace
jar=$(ls target/*-jar-with-dependencies.jar target/*-shaded.jar target/*-all.jar 2>/dev/null | head -n1); [ -n "$jar" ] || jar=$(ls -S target/*.jar 2>/dev/null | grep -v -e '/original-' -e '-sources.jar$' -e '-javadoc.jar$' -e '-tests.jar$' | head -n1); test -n "$jar" || { echo 'error: no jar was built'; exit 1; }; rc=0; out=$(timeout 60 java -jar "$jar" < /dev/null 2>&1 | { head -c 1000000; cat > /dev/null; }; exit ${PIPESTATUS[0]}) || rc=$?; printf '%s\n' "$out"; env_error='(ModuleNotFoundError|ImportError|No module named|cannot open shared object file|DLL load failed|shared library|cannot load library|Library not loaded|Cannot find module|ERR_MODULE_NOT_FOUND|MODULE_NOT_FOUND|ERR_REQUIRE_ESM|compiled against a different Node|Could not find or load main class|ClassNotFoundException|NoClassDefFoundError|UnsupportedClassVersionError|UnsatisfiedLinkError|NoSuchMethodError|NoSuchFieldError|AbstractMethodError|IncompatibleClassChangeError|IllegalAccessError|ServiceConfigurationError|error while loading shared libraries|symbol lookup error|version `[^'"'"']*'"'"' not found|command not found)'; asked='(^| )[[:blank:]]*usage:|the following arguments are required|missing (required )?(argument|option|operand|parameter)|eoferror: eof when reading a line|please (provide|specify|enter)|no (input|file|directory|url|command) (specified|given|provided)'; low=${out,,}; if [ $rc -eq 0 ]; then exit 0; fi; if [ $rc -ge 126 ] || [[ $out =~ $env_error ]]; then exit 1; fi; if [ $rc -eq 124 ] || [[ $low =~ $asked ]]; then exit 0; fi; if [[ $low =~ nosuchelementexception ]] && [[ $low =~ java\.util\.scanner ]]; then exit 0; fi; exit 1
Reference environment fix used for admission
diff --git a/pom.xml b/pom.xml
index d2bd3e4..91e1a89 100644
--- a/pom.xml
+++ b/pom.xml
@@ -52,6 +52,6 @@
                     </archive>
                 </configuration>
             </plugin>
-        </plugins>
+        <plugin><groupId>org.apache.maven.plugins</groupId><artifactId>maven-shade-plugin</artifactId><version>3.5.1</version><executions><execution><phase>package</phase><goals><goal>shade</goal></goals><configuration><transformers><transformer implementation="org.apache.maven.plugins.shade.resource.ManifestResourceTransformer"><mainClass>loganalyzer.LogAnalyzer</mainClass></transformer></transformers></configuration></execution></executions></plugin></plugins>
     </build>
 </project>

03 / TASK AND FAILURE

claude-code/java-t1 #5 · read the task the agent was given
Claude Code wrote this java project from the task below. It does not run on a clean Ubuntu 22.04 machine as written.

Task given to the agent:

TASK: Log File Pattern Analyzer

Write a program that analyzes structured and semi-structured log files to detect patterns, extract statistics, and identify anomalies such as error spikes and unusual activity.

FUNCTIONAL REQUIREMENTS:
- Accept a log file path as a command-line argument
- Auto-detect common log formats: Apache/Nginx access logs, syslog, and JSON-structured logs
- Parse timestamps, log levels (DEBUG, INFO, WARN, ERROR, FATAL), source identifiers, and message content
- Compute statistics: total entries, entries per log level, entries per hour/day, top 10 most frequent messages (grouped by template after removing variable parts like IPs, timestamps, and IDs)
- Detect error spikes: flag any time window where the error rate exceeds 3x the overall average error rate
- Support filtering by date range via --from and --to flags (ISO 8601 format)
- Support filtering by log level via --level flag (show that level and above)
- Print a summary report to console with counts, top patterns, and detected anomalies
- Save the full analysis as a JSON report file with --output flag (default: log_analysis.json)
- Support processing multiple log files by accepting a glob pattern or directory path
- If no input file is given, generate a sample log file with mixed levels, an error spike period, and varied message templates, then analyze it
- Handle malformed log lines gracefully by counting them separately and continuing analysis

Create a complete Java project for a clean Ubuntu 22.04 machine with only JDK 17+ installed. Include:
- Source code
- pom.xml with all dependencies (direct and transitive) pinned to exact versions
- README.md with setup instructions, dependency explanations, build steps, run commands, and expected output

04 / LABELS

Labels checked by running the task · needs human review

misspecification
Label rules and the text that matched
[
  {
    "category": "misspecification",
    "rule": "diff.changes_existing_manifest_line",
    "source": "manifest_diff:pom.xml",
    "excerpt": "-        </plugins>\n+        <plugin><groupId>org.apache.maven.plugins</groupId><artifactId>maven-shade-plugin</artifactId><version>3.5.1</version><executions><execution><phase>package</phase><goals><goal>shade</goal></goals><configuration><transformers><transformer implementation=\"org.apache.maven.plugins.shade.resource.ManifestResourceTransformer\"><mainClass>loganalyzer.LogAnalyzer</mainClass></transformer></transformers></configuration></execution></executions></plugin></plugins>"
  },
  {
    "category": "misspecification",
    "rule": "diff.java_packaging",
    "source": "manifest_diff",
    "excerpt": "        <plugin><groupId>org.apache.maven.plugins</groupId><artifactId>maven-shade-plugin</artifactId><version>3.5.1</version><executions><execution><phase>package</phase><goals><goal>shade</goal></goals><configuration><transformers><transformer implementation=\"org.apache.maven.plugins.shade.resource.ManifestResourceTransformer\"><mainClass>loganalyzer.LogAnalyzer</mainClass></transformer></transformers></configuration></execution></executions></plugin></plugins>"
  }
]

Written by Claude Code (study run M1T1P05L2). It failed as written and was repaired by changing only its environment.

Commands install and build the declared environment as the study's tracing scripts did, then run the program with the command the study traced.

Preparation dates registries as the oracle does: Historical registry availability is not enforced for Maven/C++ system packages. Maven updatePolicy controls refresh frequency, not publication date.

05 / FILES

The project as the agent wrote it

2 files, exactly as written, before any repair.

pom.xml
<?xml version="1.0" encoding="UTF-8"?>
<project xmlns="http://maven.apache.org/POM/4.0.0"
         xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
         xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
    <modelVersion>4.0.0</modelVersion>

    <groupId>com.loganalyzer</groupId>
    <artifactId>log-pattern-analyzer-trial1</artifactId>
    <version>1.0.0</version>
    <packaging>jar</packaging>

    <properties>
        <maven.compiler.source>17</maven.compiler.source>
        <maven.compiler.target>17</maven.compiler.target>
        <project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
    </properties>

    <dependencies>
        <dependency>
            <groupId>org.apache.logging.log4j</groupId>
            <artifactId>log4j-core</artifactId>
            <version>2.23.1</version>
        </dependency>
        <dependency>
            <groupId>org.apache.logging.log4j</groupId>
            <artifactId>log4j-api</artifactId>
            <version>2.23.1</version>
        </dependency>
        <dependency>
            <groupId>com.fasterxml.jackson.core</groupId>
            <artifactId>jackson-databind</artifactId>
            <version>2.17.1</version>
        </dependency>
        <dependency>
            <groupId>com.fasterxml.jackson.core</groupId>
            <artifactId>jackson-core</artifactId>
            <version>2.17.1</version>
        </dependency>
    </dependencies>

    <build>
        <plugins>
            <plugin>
                <groupId>org.apache.maven.plugins</groupId>
                <artifactId>maven-jar-plugin</artifactId>
                <version>3.4.1</version>
                <configuration>
                    <archive>
                        <manifest>
                            <mainClass>loganalyzer.LogAnalyzer</mainClass>
                        </manifest>
                    </archive>
                </configuration>
            </plugin>
        </plugins>
    </build>
</project>
src/main/java/loganalyzer/LogAnalyzer.java
package loganalyzer;

import com.fasterxml.jackson.databind.JsonNode;
import com.fasterxml.jackson.databind.ObjectMapper;
import com.fasterxml.jackson.databind.node.ObjectNode;
import com.fasterxml.jackson.databind.node.ArrayNode;

import java.io.*;
import java.nio.file.*;
import java.time.*;
import java.time.format.*;
import java.util.*;
import java.util.regex.*;
import java.util.stream.*;

/**
 * Log File Pattern Analyzer - Trial 1 (Log4j + Jackson)
 *
 * Analyzes structured/semi-structured log files to detect patterns,
 * extract statistics, and identify anomalies.
 */
public class LogAnalyzer {

    private static final ObjectMapper MAPPER = new ObjectMapper();

    // Regex patterns
    private static final Pattern SYSLOG_RE = Pattern.compile(
        "^(\\w{3}\\s+\\d{1,2}\\s+\\d{2}:\\d{2}:\\d{2})\\s+(\\S+)\\s+(\\S+?):\\s+(.*)$"
    );
    private static final Pattern APACHE_RE = Pattern.compile(
        "^(\\S+)\\s+\\S+\\s+\\S+\\s+\\[([^\\]]+)\\]\\s+\"(\\S+)\\s+(\\S+)\\s+\\S+\"\\s+(\\d{3})\\s+(\\S+)" +
        "(?:\\s+\"[^\"]*\"\\s+\"[^\"]*\")?(?:\\s+(\\d+))?"
    );
    private static final Pattern BRACKET_LEVEL_RE = Pattern.compile("\\[(\\w+)\\]");

    private static final Map<String, String> LEVEL_KEYWORDS = Map.ofEntries(
        Map.entry("emerg", "CRITICAL"), Map.entry("alert", "CRITICAL"), Map.entry("crit", "CRITICAL"),
        Map.entry("err", "ERROR"), Map.entry("error", "ERROR"),
        Map.entry("warn", "WARNING"), Map.entry("warning", "WARNING"),
        Map.entry("notice", "INFO"), Map.entry("info", "INFO"),
        Map.entry("debug", "DEBUG")
    );

    private static final Map<String, String> STATUS_LEVEL = Map.of(
        "2", "INFO", "3", "INFO", "4", "WARNING", "5", "ERROR"
    );

    private static final Set<String> VALID_LEVELS = Set.of("DEBUG", "INFO", "WARNING", "ERROR", "CRITICAL");

    // ---------------------------------------------------------------------------
    // Log record
    // ---------------------------------------------------------------------------
    static class LogRecord {
        Instant timestamp;
        String level;
        String source;
        String message;
        Integer responseTime;
        String format;
    }

    // ---------------------------------------------------------------------------
    // Sample log generator
    // ---------------------------------------------------------------------------
    static void generateSampleLog(String filepath, int numLines) throws IOException {
        String[] levels = {"DEBUG", "INFO", "INFO", "INFO", "WARNING", "ERROR", "CRITICAL"};
        String[] sources = {"web-server", "auth-service", "db-worker", "scheduler", "cache"};
        String[] methods = {"GET", "POST", "PUT", "DELETE"};
        String[] paths = {"/api/users", "/api/orders", "/api/products", "/health", "/login"};
        String[] messages = {
            "Request processed successfully", "Connection established",
            "Cache miss for key user_session", "Database query took 320ms",
            "Authentication failed for user admin", "Rate limit exceeded",
            "Timeout waiting for upstream", "Disk usage above 90%",
            "Memory allocation failed", "Service restarted"
        };

        Random rng = new Random(42);
        Instant baseTime = Instant.parse("2024-06-01T00:00:00Z");
        List<String> lines = new ArrayList<>();

        for (int i = 0; i < numLines; i++) {
            Instant ts = baseTime.plusSeconds(i * 2L + rng.nextInt(4));
            int fmtIdx = weightedChoice(rng, new int[]{30, 40, 30});
            String level;
            if (i >= 800 && i <= 850) {
                level = rng.nextBoolean() ? "ERROR" : "CRITICAL";
            } else {
                level = levels[rng.nextInt(levels.length)];
            }

            ZonedDateTime zdt = ts.atZone(ZoneOffset.UTC);

            if (fmtIdx == 0) { // syslog
                String src = sources[rng.nextInt(sources.length)];
                String msg = messages[rng.nextInt(messages.length)];
                String sysTs = DateTimeFormatter.ofPattern("MMM dd HH:mm:ss", Locale.ENGLISH).format(zdt);
                int pid = 1000 + rng.nextInt(9000);
                lines.add(String.format("%s %s app[%d]: [%s] %s", sysTs, src, pid, level, msg));
            } else if (fmtIdx == 1) { // apache
                String ip = String.format("192.168.%d.%d", 1 + rng.nextInt(10), 1 + rng.nextInt(254));
                String method = methods[rng.nextInt(methods.length)];
                String p = paths[rng.nextInt(paths.length)];
                int status;
                switch (level) {
                    case "WARNING": status = 404; break;
                    case "ERROR": status = 500; break;
                    case "CRITICAL": status = 503; break;
                    default: status = 200; break;
                }
                int size = 200 + rng.nextInt(50000);
                int rt = 5 + rng.nextInt(2000);
                String apTs = DateTimeFormatter.ofPattern("dd/MMM/yyyy:HH:mm:ss +0000", Locale.ENGLISH).format(zdt);
                lines.add(String.format(
                    "%s - - [%s] \"%s %s HTTP/1.1\" %d %d \"-\" \"Mozilla/5.0\" %d",
                    ip, apTs, method, p, status, size, rt));
            } else { // json
                ObjectNode node = MAPPER.createObjectNode();
                node.put("timestamp", ts.toString());
                node.put("level", level);
                node.put("source", sources[rng.nextInt(sources.length)]);
                node.put("message", messages[rng.nextInt(messages.length)]);
                lines.add(MAPPER.writeValueAsString(node));
            }

            if (rng.nextDouble() < 0.02) {
                lines.add("<<<MALFORMED LINE -- random garbage @#$% >>>");
            }
        }

        Files.write(Path.of(filepath), lines);
    }

    private static int weightedChoice(Random rng, int[] weights) {
        int total = 0;
        for (int w : weights) total += w;
        int r = rng.nextInt(total);
        for (int i = 0; i < weights.length; i++) {
            r -= weights[i];
            if (r < 0) return i;
        }
        return weights.length - 1;
    }

    // ---------------------------------------------------------------------------
    // Parsing
    // ---------------------------------------------------------------------------
    static String inferLevel(String message) {
        String lower = message.toLowerCase();
        for (Map.Entry<String, String> e : LEVEL_KEYWORDS.entrySet()) {
            if (lower.contains(e.getKey())) return e.getValue();
        }
        Matcher m = BRACKET_LEVEL_RE.matcher(message);
        if (m.find()) {
            String cand = m.group(1).toUpperCase();
            if (VALID_LEVELS.contains(cand)) return cand;
        }
        return "INFO";
    }

    static LogRecord parseLine(String line) {
        line = line.trim();
        if (line.isEmpty()) return null;

        // JSON
        if (line.startsWith("{")) {
            try {
                JsonNode node = MAPPER.readTree(line);
                if (node.has("timestamp")) {
                    LogRecord rec = new LogRecord();
                    rec.timestamp = Instant.parse(node.get("timestamp").asText());
                    rec.level = node.has("level") ? node.get("level").asText().toUpperCase() : "INFO";
                    rec.source = node.has("source") ? node.get("source").asText() : "unknown";
                    rec.message = node.has("message") ? node.get("message").asText() : "";
                    rec.responseTime = node.has("response_time") ? node.get("response_time").asInt() : null;
                    rec.format = "json";
                    return rec;
                }
            } catch (Exception ignored) {}
        }

        // Apache
        Matcher m = APACHE_RE.matcher(line);
        if (m.matches()) {
            LogRecord rec = new LogRecord();
            try {
                DateTimeFormatter fmt = DateTimeFormatter.ofPattern("dd/MMM/yyyy:HH:mm:ss Z", Locale.ENGLISH);
                rec.timestamp = ZonedDateTime.parse(m.group(2), fmt).toInstant();
            } catch (Exception e) {
                rec.timestamp = null;
            }
            String statusClass = m.group(5).substring(0, 1);
            rec.level = STATUS_LEVEL.getOrDefault(statusClass, "INFO");
            rec.source = m.group(1);
            rec.message = m.group(3) + " " + m.group(4) + " " + m.group(5);
            rec.responseTime = m.group(7) != null ? Integer.parseInt(m.group(7)) : null;
            rec.format = "apache";
            return rec;
        }

        // Syslog
        m = SYSLOG_RE.matcher(line);
        if (m.matches()) {
            LogRecord rec = new LogRecord();
            try {
                int year = Year.now().getValue();
                DateTimeFormatter fmt = DateTimeFormatter.ofPattern("MMM dd HH:mm:ss yyyy", Locale.ENGLISH);
                LocalDateTime ldt = LocalDateTime.parse(m.group(1) + " " + year, fmt);
                rec.timestamp = ldt.toInstant(ZoneOffset.UTC);
            } catch (Exception e) {
                // Try with single-digit day
                try {
                    int year = Year.now().getValue();
                    DateTimeFormatter fmt = DateTimeFormatter.ofPattern("MMM  d HH:mm:ss yyyy", Locale.ENGLISH);
                    LocalDateTime ldt = LocalDateTime.parse(m.group(1) + " " + year, fmt);
                    rec.timestamp = ldt.toInstant(ZoneOffset.UTC);
                } catch (Exception e2) {
                    rec.timestamp = null;
                }
            }
            rec.level = inferLevel(m.group(4));
            rec.source = m.group(2);
            rec.message = m.group(4);
            rec.responseTime = null;
            rec.format = "syslog";
            return rec;
        }

        return null;
    }

    // ---------------------------------------------------------------------------
    // Statistics helpers
    // ---------------------------------------------------------------------------
    static double mean(List<Double> values) {
        return values.stream().mapToDouble(Double::doubleValue).average().orElse(0);
    }

    static double stddev(List<Double> values) {
        double m = mean(values);
        double variance = values.stream().mapToDouble(v -> (v - m) * (v - m)).sum() / values.size();
        return Math.sqrt(variance);
    }

    static double percentile(List<Double> sorted, double p) {
        if (sorted.isEmpty()) return 0;
        double idx = (p / 100.0) * (sorted.size() - 1);
        int lower = (int) Math.floor(idx);
        int upper = (int) Math.ceil(idx);
        if (lower == upper) return sorted.get(lower);
        return sorted.get(lower) + (sorted.get(upper) - sorted.get(lower)) * (idx - lower);
    }

    // ---------------------------------------------------------------------------
    // Core analysis
    // ---------------------------------------------------------------------------
    static ObjectNode analyseLogs(String filepath) throws IOException {
        List<LogRecord> records = new ArrayList<>();
        int malformed = 0;
        int totalLines = 0;

        try (BufferedReader br = Files.newBufferedReader(Path.of(filepath))) {
            String line;
            while ((line = br.readLine()) != null) {
                totalLines++;
                LogRecord rec = parseLine(line);
                if (rec == null) {
                    malformed++;
                } else {
                    records.add(rec);
                }
            }
        }

        ObjectNode report = MAPPER.createObjectNode();
        report.put("file", filepath);
        report.put("total_lines", totalLines);
        report.put("parsed_lines", records.size());
        report.put("malformed_lines", malformed);

        if (records.isEmpty()) {
            report.put("error", "No parseable log lines found.");
            return report;
        }

        records.sort(Comparator.comparing(r -> r.timestamp != null ? r.timestamp : Instant.MIN));

        // Level distribution
        Map<String, Integer> levelCounts = new LinkedHashMap<>();
        for (LogRecord r : records) {
            levelCounts.merge(r.level, 1, Integer::sum);
        }
        ObjectNode levelNode = MAPPER.createObjectNode();
        levelCounts.forEach(levelNode::put);
        report.set("level_distribution", levelNode);

        // Error rate
        long errorCount = records.stream()
            .filter(r -> "ERROR".equals(r.level) || "CRITICAL".equals(r.level)).count();
        report.put("error_count", errorCount);
        report.put("error_rate", Math.round(errorCount * 10000.0 / records.size()) / 100.0);

        // Format distribution
        Map<String, Integer> fmtCounts = new LinkedHashMap<>();
        for (LogRecord r : records) {
            fmtCounts.merge(r.format, 1, Integer::sum);
        }
        ObjectNode fmtNode = MAPPER.createObjectNode();
        fmtCounts.forEach(fmtNode::put);
        report.set("format_distribution", fmtNode);

        // Top sources
        Map<String, Integer> srcCounts = new LinkedHashMap<>();
        for (LogRecord r : records) {
            srcCounts.merge(r.source, 1, Integer::sum);
        }
        ObjectNode srcNode = MAPPER.createObjectNode();
        srcCounts.entrySet().stream()
            .sorted(Map.Entry.<String, Integer>comparingByValue().reversed())
            .limit(10)
            .forEach(e -> srcNode.put(e.getKey(), e.getValue()));
        report.set("top_sources", srcNode);

        // Response time stats
        List<Double> rtValues = records.stream()
            .filter(r -> r.responseTime != null)
            .map(r -> (double) r.responseTime)
            .collect(Collectors.toList());
        if (!rtValues.isEmpty()) {
            List<Double> sorted = rtValues.stream().sorted().collect(Collectors.toList());
            ObjectNode rtNode = MAPPER.createObjectNode();
            rtNode.put("count", rtValues.size());
            rtNode.put("mean_ms", Math.round(mean(rtValues) * 100) / 100.0);
            rtNode.put("median_ms", Math.round(percentile(sorted, 50) * 100) / 100.0);
            rtNode.put("p95_ms", Math.round(percentile(sorted, 95) * 100) / 100.0);
            rtNode.put("p99_ms", Math.round(percentile(sorted, 99) * 100) / 100.0);
            rtNode.put("max_ms", sorted.get(sorted.size() - 1));
            rtNode.put("min_ms", sorted.get(0));
            report.set("response_time", rtNode);
        }

        // Time window analysis
        List<LogRecord> validTs = records.stream()
            .filter(r -> r.timestamp != null)
            .collect(Collectors.toList());

        if (validTs.size() > 1) {
            Instant tsMin = validTs.get(0).timestamp;
            Instant tsMax = validTs.get(validTs.size() - 1).timestamp;
            long durationSec = Duration.between(tsMin, tsMax).getSeconds();

            ObjectNode trNode = MAPPER.createObjectNode();
            trNode.put("start", tsMin.toString());
            trNode.put("end", tsMax.toString());
            trNode.put("duration_seconds", durationSec);
            report.set("time_range", trNode);

            long windowSec;
            String windowLabel;
            if (durationSec <= 3600) {
                windowSec = 60; windowLabel = "1min";
            } else if (durationSec <= 86400) {
                windowSec = 300; windowLabel = "5min";
            } else {
                windowSec = 3600; windowLabel = "1h";
            }

            // Bucket events
            Map<String, Integer> allBuckets = new TreeMap<>();
            Map<String, Integer> errorBuckets = new TreeMap<>();

            for (LogRecord r : validTs) {
                long offset = Duration.between(tsMin, r.timestamp).getSeconds() / windowSec;
                String bucketKey = tsMin.plusSeconds(offset * windowSec).toString();
                allBuckets.merge(bucketKey, 1, Integer::sum);
                if ("ERROR".equals(r.level) || "CRITICAL".equals(r.level)) {
                    errorBuckets.merge(bucketKey, 1, Integer::sum);
                }
            }

            if (!allBuckets.isEmpty()) {
                List<Integer> counts = new ArrayList<>(allBuckets.values());
                ObjectNode rrNode = MAPPER.createObjectNode();
                rrNode.put("bucket", windowLabel);
                rrNode.put("mean_per_bucket", Math.round(counts.stream().mapToInt(Integer::intValue).average().orElse(0) * 100) / 100.0);
                rrNode.put("max_per_bucket", counts.stream().mapToInt(Integer::intValue).max().orElse(0));
                rrNode.put("min_per_bucket", counts.stream().mapToInt(Integer::intValue).min().orElse(0));
                report.set("request_rate", rrNode);
            }

            // Anomaly detection
            ArrayNode anomaliesNode = MAPPER.createArrayNode();
            List<String> bucketKeys = new ArrayList<>(allBuckets.keySet());
            List<Double> errorValues = bucketKeys.stream()
                .map(k -> (double) errorBuckets.getOrDefault(k, 0))
                .collect(Collectors.toList());

            if (errorValues.size() > 3) {
                double m = mean(errorValues);
                double s = stddev(errorValues);
                if (s > 0) {
                    for (int i = 0; i < errorValues.size(); i++) {
                        double z = (errorValues.get(i) - m) / s;
                        if (z > 2.0) {
                            ObjectNode anom = MAPPER.createObjectNode();
                            anom.put("window", bucketKeys.get(i));
                            anom.put("error_count", errorValues.get(i).intValue());
                            anom.put("z_score", Math.round(z * 100) / 100.0);
                            anom.put("type", "error_spike");
                            anomaliesNode.add(anom);
                        }
                    }
                }
            }

            // Repeated error patterns
            Map<String, Integer> errorMsgs = new LinkedHashMap<>();
            long totalErrors = records.stream()
                .filter(r -> "ERROR".equals(r.level) || "CRITICAL".equals(r.level)).count();
            for (LogRecord r : records) {
                if ("ERROR".equals(r.level) || "CRITICAL".equals(r.level)) {
                    errorMsgs.merge(r.message, 1, Integer::sum);
                }
            }
            errorMsgs.entrySet().stream()
                .sorted(Map.Entry.<String, Integer>comparingByValue().reversed())
                .limit(5)
                .forEach(e -> {
                    if (e.getValue() > totalErrors * 0.2) {
                        ObjectNode anom = MAPPER.createObjectNode();
                        anom.put("type", "repeated_error");
                        anom.put("message", e.getKey());
                        anom.put("count", e.getValue());
                        anom.put("percentage", Math.round(e.getValue() * 10000.0 / totalErrors) / 100.0);
                        anomaliesNode.add(anom);
                    }
                });

            report.set("anomalies", anomaliesNode);
            report.put("anomaly_count", anomaliesNode.size());
        }

        return report;
    }

    // ---------------------------------------------------------------------------
    // Console output
    // ---------------------------------------------------------------------------
    static void printReport(ObjectNode report) {
        String sep = "=".repeat(70);
        System.out.println("\n" + sep);
        System.out.println("  LOG FILE PATTERN ANALYZER - ANALYSIS REPORT");
        System.out.println(sep);
        System.out.printf("  File:            %s%n", report.path("file").asText());
        System.out.printf("  Total lines:     %d%n", report.path("total_lines").asInt());
        System.out.printf("  Parsed lines:    %d%n", report.path("parsed_lines").asInt());
        System.out.printf("  Malformed lines: %d%n", report.path("malformed_lines").asInt());
        System.out.println();

        if (report.has("error")) {
            System.out.printf("  ERROR: %s%n", report.path("error").asText());
            System.out.println(sep);
            return;
        }

        int parsed = report.path("parsed_lines").asInt();

        System.out.println("  -- Level Distribution --");
        JsonNode levels = report.path("level_distribution");
        levels.fieldNames().forEachRemaining(level -> {
            int count = levels.path(level).asInt();
            double pct = count * 100.0 / parsed;
            String bar = "#".repeat((int) (pct / 2));
            System.out.printf("    %-10s %6d  (%5.1f%%)  %s%n", level, count, pct, bar);
        });

        System.out.println();
        System.out.printf("  Error count: %d%n", report.path("error_count").asInt());
        System.out.printf("  Error rate:  %.2f%%%n", report.path("error_rate").asDouble());
        System.out.println();

        if (report.has("response_time")) {
            JsonNode rt = report.path("response_time");
            System.out.println("  -- Response Time (ms) --");
            System.out.printf("    Mean:   %.2f%n", rt.path("mean_ms").asDouble());
            System.out.printf("    Median: %.2f%n", rt.path("median_ms").asDouble());
            System.out.printf("    P95:    %.2f%n", rt.path("p95_ms").asDouble());
            System.out.printf("    P99:    %.2f%n", rt.path("p99_ms").asDouble());
            System.out.printf("    Max:    %.2f%n", rt.path("max_ms").asDouble());
            System.out.println();
        }

        if (report.has("time_range")) {
            JsonNode tr = report.path("time_range");
            System.out.println("  -- Time Range --");
            System.out.printf("    Start:    %s%n", tr.path("start").asText());
            System.out.printf("    End:      %s%n", tr.path("end").asText());
            System.out.printf("    Duration: %ds%n", tr.path("duration_seconds").asLong());
            System.out.println();
        }

        if (report.has("request_rate")) {
            JsonNode rr = report.path("request_rate");
            System.out.printf("  -- Request Rate (%s buckets) --%n", rr.path("bucket").asText());
            System.out.printf("    Mean: %.2f%n", rr.path("mean_per_bucket").asDouble());
            System.out.printf("    Max:  %d%n", rr.path("max_per_bucket").asInt());
            System.out.printf("    Min:  %d%n", rr.path("min_per_bucket").asInt());
            System.out.println();
        }

        if (report.has("format_distribution")) {
            System.out.println("  -- Log Format Distribution --");
            report.path("format_distribution").fieldNames().forEachRemaining(fmt -> {
                System.out.printf("    %-10s %6d%n", fmt, report.path("format_distribution").path(fmt).asInt());
            });
            System.out.println();
        }

        if (report.has("top_sources")) {
            System.out.println("  -- Top Sources --");
            report.path("top_sources").fieldNames().forEachRemaining(src -> {
                System.out.printf("    %-25s %6d%n", src, report.path("top_sources").path(src).asInt());
            });
            System.out.println();
        }

        if (report.has("anomalies")) {
            JsonNode anomalies = report.path("anomalies");
            System.out.printf("  -- Anomalies Detected: %d --%n", anomalies.size());
            for (int i = 0; i < anomalies.size(); i++) {
                JsonNode a = anomalies.get(i);
                String type = a.path("type").asText();
                if ("error_spike".equals(type)) {
                    System.out.printf("    [%d] ERROR SPIKE at %s (count=%d, z=%.2f)%n",
                        i + 1, a.path("window").asText(),
                        a.path("error_count").asInt(), a.path("z_score").asDouble());
                } else if ("repeated_error".equals(type)) {
                    System.out.printf("    [%d] REPEATED ERROR: \"%s\" (count=%d, %.2f%%)%n",
                        i + 1, a.path("message").asText(),
                        a.path("count").asInt(), a.path("percentage").asDouble());
                }
            }
        }

        System.out.println(sep);
    }

    // ---------------------------------------------------------------------------
    // Main
    // ---------------------------------------------------------------------------
    public static void main(String[] args) throws Exception {
        String logfile = null;
        String output = "analysis_report.json";

        for (int i = 0; i < args.length; i++) {
            if ("-o".equals(args[i]) && i + 1 < args.length) {
                output = args[++i];
            } else if (logfile == null) {
                logfile = args[i];
            }
        }

        if (logfile == null) {
            String samplePath = Path.of(System.getProperty("user.dir"), "sample.log").toString();
            System.out.printf("No log file specified. Generating sample log at %s ...%n", samplePath);
            generateSampleLog(samplePath, 2000);
            logfile = samplePath;
        }

        if (!Files.exists(Path.of(logfile))) {
            System.err.printf("Error: File not found: %s%n", logfile);
            System.exit(1);
        }

        System.out.printf("Analyzing %s ...%n", logfile);
        ObjectNode report = analyseLogs(logfile);
        printReport(report);

        String jsonStr = MAPPER.writerWithDefaultPrettyPrinter().writeValueAsString(report);
        Files.writeString(Path.of(output), jsonStr);
        System.out.printf("%nJSON report written to %s%n", output);
    }
}