#!/usr/bin/env python3
"""Build aggregate SMART coverage CSV and SVG from the validated audit JSON.

Usage: python3 build_smart_coverage.py audit.json --output-dir .
No network or device access; Python standard library only.
"""
import argparse
import csv
import json
from pathlib import Path

LABELS = {
    "complete_positive": "All 5 reported, at least one positive",
    "partial_positive": "1-4 reported, at least one positive",
    "complete_zero": "All 5 reported, all zero",
    "partial_zero": "1-4 reported, reported values zero",
    "no_values": "None of the 5 reported",
}


def build(report, directory):
    directory.mkdir(parents=True, exist_ok=True)
    rows = []
    for maker in ("ALL", "Seagate", "HGST", "Toshiba", "WDC", "Other"):
        for group in ("failed", "survivor"):
            n = report["counts"].get(f"{maker}|{group}|rows", 0)
            if not n:
                continue
            counts = [report["coverage"].get(f"{maker}|{group}|{state}", 0) for state in LABELS]
            if any(type(value) is not int or value < 0 for value in counts) or sum(counts) != n:
                raise ValueError("Coverage states must partition every cohort")
            for (state, label), count in zip(LABELS.items(), counts):
                rows.append([maker, group, state, count, n, f"{100 * count / n:.6f}"])
    with (directory / "backblaze-q4-2025-smart-coverage.csv").open("w", newline="") as handle:
        writer = csv.writer(handle, lineterminator="\n")
        writer.writerow(["manufacturer", "cohort", "coverage_state", "drives", "cohort_drives", "percent"])
        writer.writerows(rows)
    failed = [row for row in rows if row[0] == "ALL" and row[1] == "failed"]
    n = report["counts"]["ALL|failed|rows"]
    left, span = 290, 390
    colors = ("#176b73", "#347fa0", "#585858", "#a45218", "#783d61")
    svg = [
        '<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 800 380" width="800" height="380" role="img" aria-labelledby="title desc">',
        '<title id="title">SMART coverage on the recorded failure day, Backblaze Q4 2025</title>',
        f'<desc id="desc">Five mutually exclusive groups partition {n} capacity-selected failed drives. Exact counts are in the adjacent table and downloadable CSV.</desc>',
        '<rect width="800" height="380" fill="#ffffff"/>',
        '<g font-family="sans-serif" fill="#222222">',
        '<text x="20" y="30" font-size="18" font-weight="bold">A missing value is not a measured zero</text>',
        f'<text x="20" y="53" font-size="13">Q4 2025 failure-day rows; n = {n}; capacity at least 3 TB</text>',
    ]
    for tick in range(0, 101, 20):
        x = left + span * tick / 100
        svg.append(f'<line x1="{x}" y1="79" x2="{x}" y2="294" stroke="#dddddd"/>')
        svg.append(f'<text x="{x}" y="319" text-anchor="middle" font-size="12">{tick}%</text>')
    for i, row in enumerate(failed):
        _, _, state, count, _, percent = row
        y = 91 + i * 43
        width = span * count / n
        svg.append(f'<text x="20" y="{y+16}" font-size="13">{LABELS[state]}</text>')
        svg.append(f'<rect x="{left}" y="{y}" width="{width:.3f}" height="24" fill="{colors[i]}"/>')
        svg.append(f'<text x="{left+width+8:.3f}" y="{y+16}" font-size="13">{count} ({float(percent):.1f}%)</text>')
    svg.extend([
        '<text x="290" y="343" font-size="12">Share of all selected failed drives (0-100% scale)</text>',
        '<text x="20" y="368" font-size="12">Source: Backblaze daily files; extraction: Hesela. Descriptive counts, not prediction accuracy.</text>',
        '</g></svg>',
    ])
    (directory / "backblaze-q4-2025-smart-coverage.svg").write_text("\n".join(svg) + "\n")
    return rows


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("audit", type=Path)
    parser.add_argument("--output-dir", type=Path, required=True)
    args = parser.parse_args()
    rows = build(json.loads(args.audit.read_text()), args.output_dir)
    print(f"Generated {len(rows)} coverage rows and a count-derived chart")
