#!/usr/bin/env python3
"""Audit Q4 2025 SMART availability and reproduce the existing failure-day table.

Python 3.11+, standard library only. Reads an archive; no device or network access.
Public output contains aggregates, not serial numbers or infrastructure identifiers.
"""
import argparse
from collections import Counter
import csv
from datetime import date, timedelta, datetime, timezone
import hashlib
import io
import json
import math
from pathlib import Path
import platform
import zipfile

ATTRIBUTES = (5, 187, 188, 197, 198)
STATES = ("complete_positive", "partial_positive", "complete_zero", "partial_zero", "no_values")
SOURCE = "https://f001.backblazeb2.com/file/Backblaze-Hard-Drive-Data/data_Q4_2025.zip"


def number(raw):
    try:
        value = float(raw)
    except (ValueError, TypeError):
        return None
    return value if math.isfinite(value) and value >= 0 else None


def manufacturer(model):
    model = model.upper()
    if model.startswith("ST"):
        return "Seagate"
    if model.startswith(("HGST", "HUH", "HMS")):
        return "HGST"
    if model.startswith("TOSHIBA"):
        return "Toshiba"
    if model.startswith(("WDC", "WUH", "WD")):
        return "WDC"
    return "Other"


def classify(raw_values):
    values = [number(raw) for raw in raw_values]
    valid = sum(value is not None for value in values)
    positive = any(value is not None and value > 0 for value in values)
    if not valid:
        state = "no_values"
    else:
        state = ("complete_" if valid == len(ATTRIBUTES) else "partial_") + ("positive" if positive else "zero")
    return values, state


def add(counts, coverage, group, maker, raw_values):
    values, state = classify(raw_values)
    mask = "".join("1" if value is not None else "0" for value in values)
    for m in ("ALL", maker):
        counts[f"{m}|{group}|rows"] += 1
        coverage[f"{m}|{group}|{state}"] += 1
        coverage[f"{m}|{group}|mask|{mask}"] += 1
        for attr, raw, value in zip(ATTRIBUTES, raw_values, values):
            if value is None:
                reason = "missing" if not raw.strip() else "invalid"
                coverage[f"{m}|{group}|{attr}|{reason}"] += 1
            else:
                counts[f"{m}|{group}|{attr}|filled"] += 1
                counts[f"{m}|{group}|{attr}|pos"] += value > 0
        counts[f"{m}|{group}|any|filled"] += state != "no_values"
        counts[f"{m}|{group}|any|pos"] += state.endswith("_positive")


def sha256(path):
    with open(path, "rb") as handle:
        return hashlib.file_digest(handle, "sha256").hexdigest()


def aggregate(path):
    counts, coverage, totals = Counter(), Counter(), Counter()
    failed_serials, survivor_serials = set(), set()
    expected = [(date(2025, 10, 1) + timedelta(days=i)).isoformat() for i in range(92)]
    with zipfile.ZipFile(path) as archive:
        names = sorted(n for n in archive.namelist() if n.endswith(".csv"))
        if [Path(n).stem for n in names] != expected:
            raise ValueError("Expected exactly one daily CSV for every Q4 2025 date")
        for name in names:
            day = Path(name).stem
            last_day = day == "2025-12-31"
            with archive.open(name) as handle:
                reader = csv.reader(io.TextIOWrapper(handle, encoding="utf-8", newline=""))
                header = next(reader, [])
                fields = ("date", "serial_number", "model", "capacity_bytes", "failure")
                required = fields + tuple(f"smart_{a}_raw" for a in ATTRIBUTES)
                if any(header.count(field) != 1 for field in required):
                    raise ValueError(f"{name}: missing or duplicate required columns")
                indices = [header.index(field) for field in fields]
                smart_indices = [header.index(f"smart_{a}_raw") for a in ATTRIBUTES]
                seen = set()
                for row in reader:
                    totals["raw_rows"] += 1
                    if len(row) != len(header):
                        raise ValueError(f"{name}: inconsistent row length")
                    row_day, serial, model, capacity_raw, failure = (row[i] for i in indices)
                    if row_day != day or not serial or serial in seen:
                        raise ValueError(f"{name}: wrong date, empty serial or duplicate drive-day")
                    seen.add(serial)
                    if not model.strip():
                        raise ValueError(f"{name}: empty model")
                    if failure not in ("0", "1"):
                        raise ValueError(f"{name}: failure must be 0 or 1")
                    capacity = number(capacity_raw)
                    if capacity is None:
                        totals["invalid_capacity_rows"] += 1
                        totals["invalid_capacity_failures"] += int(failure)
                        continue
                    if capacity < 3e12:
                        totals["below_3tb_rows"] += 1
                        continue
                    totals["eligible_drive_days"] += 1
                    if failure == "1":
                        if serial in failed_serials:
                            raise ValueError(f"{name}: repeated failure event for one drive")
                        failed_serials.add(serial)
                        group = "failed"
                    elif last_day:
                        survivor_serials.add(serial)
                        group = "survivor"
                    else:
                        continue
                    add(counts, coverage, group, manufacturer(model), [row[i] for i in smart_indices])
            print(f"Audited {Path(name).name}", flush=True)
    if failed_serials & survivor_serials:
        raise ValueError("Failure cohort overlaps December 31 nonfailed snapshot")
    return {
        "schema_version": 1,
        "source_url": SOURCE,
        "archive_sha256": sha256(path),
        "generator_sha256": sha256(Path(__file__)),
        "python_version": platform.python_version(),
        "generated_at": datetime.now(timezone.utc).isoformat(),
        "files": len(names), "first": names[0], "last": names[-1],
        "selection": "capacity_bytes >= 3e12; failed=all failure=1 rows in Q4; survivor=December 31 failure=0 rows; capacity alone is not a universal media classifier",
        "valid_value": "finite and nonnegative; blank is missing, nonblank rejected value is invalid",
        "attributes": list(ATTRIBUTES),
        "coverage_mask_order": list(ATTRIBUTES),
        "coverage_states": list(STATES),
        "counts": dict(sorted(counts.items())),
        "coverage": dict(sorted(coverage.items())),
        "totals": {**dict(sorted(totals.items())), "unique_failed_drives": len(failed_serials),
                   "unique_december31_nonfailed_drives": len(survivor_serials), "cohort_overlap": 0},
        "limits": [
            "Failure-day association, not advance prediction or causal inference.",
            "No-value and partial-value records cannot establish five zero attributes.",
            "Model-prefix grouping is not a device capability or vendor-encoding validation.",
            "No parsing of packed counters, history imputation or collection-failure diagnosis.",
            "December 31 nonfailed drives are a snapshot, not all nonfailed drives exposed in Q4."
        ],
    }


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("archive", type=Path)
    parser.add_argument("--output", type=Path, required=True)
    args = parser.parse_args()
    result = aggregate(args.archive)
    args.output.parent.mkdir(parents=True, exist_ok=True)
    args.output.write_text(json.dumps(result, indent=2, allow_nan=False) + "\n", encoding="utf-8")
    print(json.dumps(result["totals"], sort_keys=True))


if __name__ == "__main__":
    main()
