#!/usr/bin/env python3
"""Audit and aggregate Q4 2025 drive-day age data using only the standard library.

Usage: python audit_backblaze_age.py ARCHIVE.zip --output audit.json
No device access or network requests. The archive is streamed, not extracted.
Output contains aggregates only, never serial numbers or infrastructure identifiers.
"""
import argparse
import collections
import csv
from datetime import date, timedelta, datetime, timezone
import hashlib
import io
import json
import math
from pathlib import Path
import platform
import zipfile

SOURCE = "https://f001.backblazeb2.com/file/Backblaze-Hard-Drive-Data/data_Q4_2025.zip"
CLASSES = ("up to 12 TB", "over 12 to 16 TB", "over 16 TB")
HOURS_PER_YEAR = 8766.0


def finite_number(value):
    try:
        number = float(value)
    except (ValueError, TypeError):
        return None
    return number if math.isfinite(number) else None


def capacity_class(nominal_tb):
    if nominal_tb <= 12:
        return CLASSES[0]
    return CLASSES[1] if nominal_tb <= 16 else CLASSES[2]


def age_bucket(value):
    hours = finite_number(value)
    if hours is None or hours < 0:
        return None
    return min(int(hours / HOURS_PER_YEAR), 8)


def sha256(path):
    with open(path, "rb") as source:
        return hashlib.file_digest(source, "sha256").hexdigest()


def aggregate(archive):
    counts = collections.Counter()
    totals = collections.Counter()
    capacities = collections.defaultdict(collections.Counter)
    missing_models = collections.defaultdict(collections.Counter)
    with zipfile.ZipFile(archive) as z:
        names = sorted(n for n in z.namelist() if n.endswith(".csv"))
        expected = [(date(2025, 10, 1) + timedelta(days=i)).isoformat() for i in range(92)]
        if [Path(n).stem for n in names] != expected:
            raise ValueError("Expected exactly one daily CSV for every Q4 2025 date")
        for n in names:
            with z.open(n) as source:
                reader = csv.reader(io.TextIOWrapper(source, encoding="utf-8", newline=""))
                header = next(reader)
                fields = ("date", "serial_number", "model", "capacity_bytes", "failure", "smart_9_raw")
                if any(header.count(k) != 1 for k in fields):
                    raise ValueError(f"{n}: missing or duplicate required columns")
                ix = [header.index(k) for k in fields]
                seen = set()
                for row in reader:
                    totals["raw_rows"] += 1
                    if len(row) != len(header):
                        raise ValueError(f"{n}: inconsistent row length")
                    day, serial, model, cap_raw, failure_raw, poh = (row[i] for i in ix)
                    if day != Path(n).stem or not serial or serial in seen:
                        raise ValueError(f"{n}: wrong date, empty serial or duplicate drive-day")
                    seen.add(serial)
                    if failure_raw not in ("0", "1"):
                        raise ValueError(f"{n}: failure must be 0 or 1")
                    failure = int(failure_raw)
                    cap = finite_number(cap_raw)
                    if cap is None or cap < 0:
                        totals["invalid_capacity_rows"] += 1
                        totals["invalid_capacity_failures"] += failure
                        continue
                    if cap < 3e12:
                        totals["below_3tb_rows"] += 1
                        totals["below_3tb_failures"] += failure
                        continue
                    totals["eligible_drive_days"] += 1
                    totals["eligible_failures"] += failure
                    tb = round(cap / 1e12)
                    capacities[tb]["drive_days"] += 1
                    capacities[tb]["failures"] += failure
                    bucket = age_bucket(poh)
                    if bucket is None:
                        reason = "missing" if not poh.strip() else "invalid_age"
                        counts[f"{reason}|days"] += 1
                        counts[f"{reason}|fail"] += failure
                        missing_models[model][f"{reason}_drive_days"] += 1
                        missing_models[model][f"{reason}_failures"] += failure
                        continue
                    totals["with_age_drive_days"] += 1
                    totals["with_age_failures"] += failure
                    capacities[tb]["with_age_drive_days"] += 1
                    capacities[tb]["with_age_failures"] += failure
                    for group in ("All", capacity_class(tb)):
                        counts[f"{group}|{bucket}|days"] += 1
                        counts[f"{group}|{bucket}|fail"] += failure
            print(f"Audited {Path(n).name}", flush=True)
    return {
        "schema_version": 1,
        "source_url": SOURCE,
        "archive_sha256": sha256(archive),
        "generator_sha256": sha256(Path(__file__)),
        "python_version": platform.python_version(),
        "generated_at": datetime.now(timezone.utc).isoformat(),
        "files": len(names), "first": names[0], "last": names[-1],
        "selection": "capacity_bytes >= 3e12 is a capacity filter, not a universal media-type test; SMART 9 must be finite and nonnegative for age buckets",
        "age_year_hours": HOURS_PER_YEAR,
        "annualization_days": 365,
        "counts": dict(sorted(counts.items())),
        "totals": dict(sorted(totals.items())),
        "capacity_counts": [{"nominal_tb": k, **dict(sorted(v.items()))} for k, v in sorted(capacities.items())],
        "excluded_age_by_model": [{"model": k, **dict(sorted(v.items()))} for k, v in sorted(missing_models.items())],
    }


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("archive", type=Path)
    parser.add_argument("--output", type=Path, required=True)
    args = parser.parse_args()
    result = aggregate(args.archive)
    args.output.parent.mkdir(parents=True, exist_ok=True)
    args.output.write_text(json.dumps(result, indent=2) + "\n", encoding="utf-8")
    print(json.dumps(result["totals"], sort_keys=True))
