#!/usr/bin/env python3
"""Reproducible OpenSSH authentication-log analysis using Python's standard library.

This tool identifies records and review candidates; it does not determine intent,
attribute a person, prove compromise, or recover activity absent from the source.
"""
from __future__ import annotations

import argparse
import csv
import hashlib
import html
import ipaddress
import json
import re
import sys
from collections import Counter, defaultdict, deque
from datetime import datetime, timedelta, timezone
from pathlib import Path
from typing import Any

VERSION = "1.0.1"
MONTHS = {m: n for n, m in enumerate(
    ("Jan", "Feb", "Mar", "Apr", "May", "Jun", "Jul", "Aug", "Sep", "Oct", "Nov", "Dec"), 1)}
BSD_PREFIX = re.compile(r"^(?P<month>[A-Z][a-z]{2})\s+(?P<day>\d{1,2})\s+(?P<clock>\d{2}:\d{2}:\d{2})\s+(?P<body>.*)$")
ISO_PREFIX = re.compile(r"^(?P<stamp>[0-9]{4}-[0-9]{2}-[0-9]{2}[Tt ][0-9]{2}:[0-9]{2}:[0-9]{2}(?:\.[0-9]+)?(?:[Zz]|[+-][0-9]{2}:[0-9]{2}))\s+(?P<body>.*)$")
ISO_WITHOUT_OFFSET = re.compile(r"^[0-9]{4}-[0-9]{2}-[0-9]{2}[Tt ][0-9]{2}:[0-9]{2}:[0-9]{2}(?:\.[0-9]+)?\s+")
RFC3339 = re.compile(r"[0-9]{4}-[0-9]{2}-[0-9]{2}[Tt][0-9]{2}:[0-9]{2}:[0-9]{2}(?:\.[0-9]+)?(?:[Zz]|(?P<sign>[+-])(?P<offset_hour>[0-9]{2}):(?P<offset_minute>[0-9]{2}))")
DAEMON = re.compile(r"\b(?P<daemon>sshd(?:-[A-Za-z0-9_-]+)?)(?:\[(?P<pid>\d+)\])?:\s*(?P<message>.*)$")
AUTH = re.compile(r"^(?P<outcome>Failed|Accepted) (?P<method>\S+) for (?P<invalid>invalid user )?(?P<user>.*?) from (?P<address>\S+) port (?P<port>\d+)(?:\s.*)?$")
INVALID = re.compile(r"^Invalid user (?P<user>.*?) from (?P<address>\S+) port (?P<port>\d+)(?:\s.*)?$")
ADDRESS_PORT = re.compile(r"(?P<address>(?:[0-9A-Fa-f:.]+)(?:%[\w.-]+)?) port (?P<port>\d+)")
SESSION = re.compile(r"\bsession (?P<action>opened|closed) for user (?P<user>[^\s(]+)")
CONNECTION_DROP = re.compile(r"^drop connection #(?P<index>[0-9]+) from \[(?P<address>[^\]]+)\]:(?P<port>[0-9]+) on \[(?P<destination>[^\]]+)\]:(?P<destination_port>[0-9]+) (?P<reason>.+)$")
PENALTY_NEW = re.compile(r"^srclimit_penalise: (?P<family>ipv4|ipv6): new (?P<network>\S+) (?P<state>deferred|active) penalty of (?P<seconds>[0-9]+(?:\.[0-9]+)?) seconds for (?P<reason>penalty: .+)$")
PENALTY_ACTIVATION = re.compile(r"^srclimit_penalise: (?P<network>\S+): activating (?P<family>ipv4|ipv6) penalty of (?P<seconds>[0-9]+(?:\.[0-9]+)?) seconds for (?P<reason>penalty: .+)$")
POLICY_FIELDS = ("source_network", "penalty_state", "penalty_seconds", "policy_reason",
                 "destination_address", "destination_port", "connection_index")
DISCONNECT_PREFIXES = ("Received disconnect from ", "Disconnected from ", "Disconnecting ",
                       "Connection closed by ", "Connection reset by ", "Closing connection to ")


class InputError(ValueError):
    """Input has an ambiguous or invalid time convention."""


def iso_utc(value: datetime | None) -> str | None:
    return None if value is None else value.astimezone(timezone.utc).isoformat().replace("+00:00", "Z")


def parse_iso(value: str) -> datetime:
    match = RFC3339.fullmatch(value) if isinstance(value, str) else None
    if not match:
        raise InputError("Timestamp must use RFC 3339 calendar date, T separator, and an explicit UTC offset or Z.")
    if match["sign"] is not None and (int(match["offset_hour"]) > 23 or int(match["offset_minute"]) > 59):
        raise InputError("Invalid RFC 3339 offset: hours must be <=23 and minutes <=59.")
    try:
        result = datetime.fromisoformat(value.replace("t", "T").replace("z", "Z").replace("Z", "+00:00"))
    except (ValueError, TypeError) as exc:
        raise InputError(f"Invalid ISO timestamp: {value!r}") from exc
    if result.tzinfo is None or result.utcoffset() is None:
        raise InputError("ISO timestamps must include an explicit UTC offset or Z.")
    return result.astimezone(timezone.utc)


def parse_timezone(value: str | None) -> timezone | None:
    if value is None:
        return None
    if value.upper() in ("UTC", "Z", "+00:00"):
        return timezone.utc
    match = re.fullmatch(r"([+-])(\d\d):(\d\d)", value)
    if not match:
        raise InputError("--timezone must be UTC or a fixed offset such as +02:00. Use an offset explicitly; local DST is not inferred.")
    hours, minutes = int(match[2]), int(match[3])
    if hours > 23 or minutes > 59:
        raise InputError("Invalid fixed timezone offset.")
    offset = timedelta(hours=hours, minutes=minutes)
    return timezone(offset if match[1] == "+" else -offset)


def native_timestamp(line: str, year: int | None, tz: timezone | None) -> tuple[datetime | None, str, str | None, int | None]:
    iso = ISO_PREFIX.match(line)
    if iso:
        return parse_iso(iso["stamp"]), iso["body"], iso["stamp"], None
    if ISO_WITHOUT_OFFSET.match(line):
        raise InputError("ISO timestamps must include an explicit UTC offset or Z.")
    bsd = BSD_PREFIX.match(line)
    if bsd and bsd["month"] in MONTHS:
        if year is None or tz is None:
            raise InputError("BSD syslog omits year and timezone: both --year and --timezone are required. No value is inferred from the current clock.")
        month = MONTHS[bsd["month"]]
        try:
            hour, minute, second = map(int, bsd["clock"].split(":"))
            stamp = datetime(year, month, int(bsd["day"]), hour, minute, second, tzinfo=tz)
        except ValueError as exc:
            raise InputError(f"Invalid BSD timestamp: {line[:16]!r}") from exc
        return stamp.astimezone(timezone.utc), bsd["body"], line[:bsd.start("body")].rstrip(), month
    return None, line, None, None


def valid_address(value: str) -> bool:
    try:
        ipaddress.ip_address(value.split("%", 1)[0])
        return True
    except ValueError:
        return False


def classify(message: str) -> dict[str, Any] | None:
    match = CONNECTION_DROP.match(message)
    if match:
        if (not valid_address(match["address"]) or not valid_address(match["destination"])
                or not 0 <= int(match["port"]) <= 65535 or not 0 <= int(match["destination_port"]) <= 65535):
            return None
        return {"kind": "connection_refused", "method": None, "user": None,
                "source_address": match["address"], "source_port": int(match["port"]), "invalid_user": False,
                "destination_address": match["destination"], "destination_port": int(match["destination_port"]),
                "connection_index": int(match["index"]), "policy_reason": match["reason"]}
    match = PENALTY_NEW.match(message) or PENALTY_ACTIVATION.match(message)
    if match:
        try:
            network = ipaddress.ip_network(match["network"], strict=False)
        except ValueError:
            return None
        if network.version != int(match["family"][-1]):
            return None
        return {"kind": "source_penalty", "method": None, "user": None, "source_address": None,
                "source_port": None, "invalid_user": False, "source_network": match["network"],
                "penalty_state": match.groupdict().get("state") or "activated",
                "penalty_seconds": match["seconds"], "policy_reason": match["reason"]}
    match = AUTH.match(message)
    if match:
        if not valid_address(match["address"]) or not 0 <= int(match["port"]) <= 65535:
            return None
        return {"kind": "authentication_failure" if match["outcome"] == "Failed" else "authentication_success",
                "method": match["method"], "user": match["user"], "source_address": match["address"],
                "source_port": int(match["port"]), "invalid_user": bool(match["invalid"])}
    match = INVALID.match(message)
    if match:
        if not valid_address(match["address"]) or not 0 <= int(match["port"]) <= 65535:
            return None
        return {"kind": "invalid_user_notice", "method": None, "user": match["user"],
                "source_address": match["address"], "source_port": int(match["port"]), "invalid_user": True}
    match = SESSION.search(message)
    if match:
        return {"kind": "session_open" if match["action"] == "opened" else "session_close",
                "method": None, "user": match["user"], "source_address": None, "source_port": None,
                "invalid_user": False}
    if message.startswith(DISCONNECT_PREFIXES):
        endpoint = ADDRESS_PORT.search(message)
        if endpoint and (not valid_address(endpoint["address"]) or not 0 <= int(endpoint["port"]) <= 65535):
            return None
        user = re.search(r"(?:authenticating user |invalid user |Disconnected from user )([^\s]+)", message)
        return {"kind": "disconnect", "method": None, "user": user[1] if user else None,
                "source_address": endpoint["address"] if endpoint else None,
                "source_port": int(endpoint["port"]) if endpoint else None,
                "invalid_user": "invalid user " in message}
    return None


def read_source(path: Path, year: int | None = None, timezone_name: str | None = None) -> dict[str, Any]:
    """Read raw syslog or a JSONL envelope without changing the source bytes."""
    data = path.read_bytes()
    raw_text = data.decode("utf-8", errors="replace")
    tz = parse_timezone(timezone_name)
    counts: Counter[str] = Counter()
    events: list[dict[str, Any]] = []
    unparsed: list[dict[str, Any]] = []
    months: set[int] = set()
    physical_lines = raw_text.split("\n")
    if physical_lines[-1] == "":
        physical_lines.pop()
    for number, physical_line in enumerate(physical_lines, 1):
        # Unicode NEL/LS/PS and VT/FF are data, not physical log boundaries.
        original = physical_line[:-1] if physical_line.endswith("\r") else physical_line
        counts["total_lines"] += 1
        if not original.strip():
            counts["blank_lines"] += 1
            continue
        line = original
        receipt = None
        scenario = None
        if original.lstrip().startswith("{"):
            try:
                envelope = json.loads(original)
                if not isinstance(envelope, dict) or not isinstance(envelope.get("line"), str):
                    raise ValueError("JSONL record must contain a string 'line'.")
                line = envelope["line"]
                scenario = envelope.get("scenario")
                if scenario is not None and not isinstance(scenario, str):
                    raise ValueError("'scenario' must be a string if present.")
                if "received_at" in envelope:
                    if not isinstance(envelope["received_at"], str):
                        raise ValueError("'received_at' must be an ISO timestamp string.")
                    receipt = parse_iso(envelope["received_at"])
                elif "timestamp" in envelope:
                    if envelope.get("timestamp_origin") != "collector_receipt":
                        raise ValueError("'timestamp' requires timestamp_origin='collector_receipt'; use received_at for unambiguous collection time.")
                    receipt = parse_iso(envelope["timestamp"])
            except (ValueError, TypeError) as exc:
                counts["malformed_json_lines"] += 1
                unparsed.append({"source_line_number": number, "source_line": original, "reason": str(exc)})
                continue
        try:
            native, body, original_timestamp, month = native_timestamp(line, year, tz)
        except InputError as exc:
            # Ambiguous time conventions are a corpus-level error, never a silent guess.
            raise InputError(f"Line {number}: {exc}") from exc
        if month is not None:
            months.add(month)
        daemon = DAEMON.search(body)
        message = daemon["message"] if daemon else body
        event = classify(message)
        if event is None:
            if daemon or re.match(r"^(?:Failed |Accepted |Invalid user |pam_)", message):
                counts["unparsed_ssh_lines"] += 1
                unparsed.append({"source_line_number": number, "source_line": original,
                                 "raw_message": line, "reason": "Unrecognized SSH message or invalid endpoint."})
            else:
                counts["unrelated_lines"] += 1
            continue
        timestamp = native if native is not None else receipt
        origin = "daemon_native" if native is not None else ("collector_receipt" if receipt is not None else "absent")
        host = None
        if daemon and original_timestamp:
            prefix = body[:daemon.start()].strip()
            host = prefix.split()[0] if prefix else None
        event.update({"event_id": f"L{number:08d}", "source_line_number": number, "source_line": original,
                      "raw_message": line, "message": message, "timestamp": iso_utc(timestamp),
                      "timestamp_origin": origin, "original_timestamp": original_timestamp,
                      "collector_received_at": iso_utc(receipt), "pid": int(daemon["pid"]) if daemon and daemon["pid"] else None,
                      "daemon": daemon["daemon"] if daemon else None, "host": host, "scenario": scenario})
        for field in POLICY_FIELDS:
            event.setdefault(field, None)
        events.append(event)
        counts["parsed_event_lines"] += 1
    if 1 in months and 12 in months:
        raise InputError("BSD corpus contains January and December: the year boundary is ambiguous. Split it into independently year-attributed files or collect ISO timestamps with years; rollover is never guessed.")
    # All UTC strings have identical timezone convention; fractional seconds can differ,
    # so use datetime rather than lexicographic order for actual event ordering.
    events.sort(key=lambda event: (event["timestamp"] is None,
                                  parse_iso(event["timestamp"]) if event["timestamp"] else datetime.max.replace(tzinfo=timezone.utc),
                                  event["source_line_number"]))
    for key in ("total_lines", "blank_lines", "parsed_event_lines", "malformed_json_lines", "unparsed_ssh_lines", "unrelated_lines"):
        counts.setdefault(key, 0)
    return {"source": {"filename": path.name, "sha256": hashlib.sha256(data).hexdigest(), "bytes": len(data),
                       "encoding": "UTF-8, replacement decoding if malformed", "decoding_replacement_characters": raw_text.count("\ufffd"),
                       "bsd_year": year, "bsd_timezone": timezone_name},
            "coverage": dict(counts), "events": events, "unparsed_records": unparsed}


def detect(events: list[dict[str, Any]], threshold: int = 5, window_seconds: int = 60) -> list[dict[str, Any]]:
    """Emit each qualifying endpoint of an inclusive rolling window, not incidents."""
    by_source: dict[str, deque[dict[str, Any]]] = defaultdict(deque)
    by_account: dict[tuple[str, str], deque[dict[str, Any]]] = defaultdict(deque)
    alerts: list[dict[str, Any]] = []
    ordered = sorted((event for event in events if event["timestamp"]),
                     key=lambda event: (parse_iso(event["timestamp"]), event["source_line_number"]))
    for event in ordered:
        address, user = event["source_address"], event["user"]
        if not address or event["kind"] not in ("authentication_failure", "authentication_success"):
            continue
        moment = parse_iso(event["timestamp"])
        cutoff = moment - timedelta(seconds=window_seconds)
        source_queue = by_source[address]
        account_queue = by_account[(address, user)]
        for queue in (source_queue, account_queue):
            while queue and parse_iso(queue[0]["timestamp"]) < cutoff:
                queue.popleft()
        if event["kind"] == "authentication_failure":
            source_queue.append(event)
            account_queue.append(event)
            if len(source_queue) >= threshold:
                alerts.append({"rule": "concentrated_failures", "classification": "review_candidate",
                               "event_id": event["event_id"], "timestamp": event["timestamp"],
                               "source_address": address, "user": None, "failure_count": len(source_queue),
                               "window_seconds": window_seconds, "threshold": threshold,
                               "first_failure_timestamp": source_queue[0]["timestamp"],
                               "evidence_event_ids": [item["event_id"] for item in source_queue]})
        elif len(account_queue) >= threshold:
            alerts.append({"rule": "success_after_failures", "classification": "review_candidate_not_proof_of_compromise",
                           "event_id": event["event_id"], "timestamp": event["timestamp"],
                           "source_address": address, "user": user, "failure_count": len(account_queue),
                           "window_seconds": window_seconds, "threshold": threshold,
                           "first_failure_timestamp": account_queue[0]["timestamp"],
                           "evidence_event_ids": [item["event_id"] for item in account_queue] + [event["event_id"]]})
    return alerts


def summarize(parsed: dict[str, Any], threshold: int = 5, window_seconds: int = 60) -> dict[str, Any]:
    if threshold < 1 or window_seconds < 1:
        raise InputError("Threshold and window must be positive integers.")
    events = parsed["events"]
    kinds = Counter(event["kind"] for event in events)
    methods = Counter(event["method"] for event in events if event["kind"] in ("authentication_failure", "authentication_success"))
    origins = Counter(event["timestamp_origin"] for event in events)
    timed = [event["timestamp"] for event in events if event["timestamp"]]
    by_source: dict[str, Counter[str]] = defaultdict(Counter)
    by_hour: dict[str, Counter[str]] = defaultdict(Counter)
    by_minute: dict[str, Counter[str]] = defaultdict(Counter)
    by_scenario: dict[str, Counter[str]] = defaultdict(Counter)
    for event in events:
        if event["source_address"]:
            by_source[event["source_address"]][event["kind"]] += 1
        if event["timestamp"]:
            hour = parse_iso(event["timestamp"]).strftime("%Y-%m-%dT%H:00:00Z")
            by_hour[hour][event["kind"]] += 1
            minute = parse_iso(event["timestamp"]).strftime("%Y-%m-%dT%H:%M:00Z")
            by_minute[minute][event["kind"]] += 1
        if event["scenario"]:
            by_scenario[event["scenario"]][event["kind"]] += 1
    sensitivity = []
    for candidate in sorted(set((3, 5, 8, threshold))):
        findings = detect(events, candidate, window_seconds)
        concentrated = [item for item in findings if item["rule"] == "concentrated_failures"]
        successes = [item for item in findings if item["rule"] == "success_after_failures"]
        sensitivity.append({"threshold": candidate, "qualifying_failure_endpoints": len(concentrated),
                            "sources_with_concentrated_failures": len({item["source_address"] for item in concentrated}),
                            "success_after_failure_review_candidates": len(successes)})
    warnings = []
    if origins["collector_receipt"]:
        warnings.append("Collector receipt times describe acquisition, not exact daemon event times; buffering and scheduling can affect window results.")
    if origins["absent"]:
        warnings.append("Untimestamped events are included in counts and placed last in the timeline; they cannot enter time-window detection.")
    if parsed["coverage"]["unparsed_ssh_lines"] or parsed["coverage"]["malformed_json_lines"]:
        warnings.append("Some records were not parsed. Review unparsed_records before treating event counts as complete.")
    if parsed["source"]["decoding_replacement_characters"]:
        warnings.append("The decoded source contains replacement characters. SHA-256 still describes the exact input bytes; decoded strings may be lossy.")
    if parsed["source"]["bsd_year"] is not None:
        warnings.append("BSD timestamps have an analyst-supplied year and fixed timezone offset; neither is established by the log bytes alone.")
    return {"schema_version": 1, "tool": {"name": "CYBERY di Mariani Yuri SSH Forensic Analyzer", "version": VERSION},
            "source": parsed["source"], "coverage": parsed["coverage"],
            "parameters": {"threshold": threshold, "window_seconds": window_seconds, "inclusive_window": True},
            "statistics": {"event_count": len(events), "counts_by_kind": dict(sorted(kinds.items())),
                           "counts_by_method": dict(sorted(methods.items())), "timestamp_origins": dict(sorted(origins.items())),
                           "first_timestamp": timed[0] if timed else None, "last_timestamp": timed[-1] if timed else None,
                           "invalid_account_failures": sum(event["invalid_user"] for event in events if event["kind"] == "authentication_failure"),
                           "by_source": {key: dict(sorted(value.items())) for key, value in sorted(by_source.items())},
                           "by_utc_hour": {key: dict(sorted(value.items())) for key, value in sorted(by_hour.items())},
                           "by_utc_minute": {key: dict(sorted(value.items())) for key, value in sorted(by_minute.items())},
                           "by_supplied_scenario": {key: dict(sorted(value.items())) for key, value in sorted(by_scenario.items())}},
            "alerts": detect(events, threshold, window_seconds), "threshold_sensitivity": sensitivity,
            "warnings": warnings, "events": events, "unparsed_records": parsed["unparsed_records"],
            "interpretation": {"authentication_counting": "Each Failed/Accepted daemon record is counted once; Invalid user and PAM notices are contextual records, not additional failed attempts.",
                               "alerts": "Qualifying rolling-window endpoints can overlap. They are review candidates, not distinct incidents or proof of compromise.",
                               "integrity": "SHA-256 verifies the bytes against this report; it does not independently establish authenticity, completeness, origin, or clock accuracy."}}


def markdown_cell(value: Any) -> str:
    if value is None:
        return "—"
    plain = str(value).replace("\r", " ").replace("\n", " ")
    escaped = re.sub(r"([\\*_{}\[\]()#+.!|~\-])", lambda match: "\\" + match[0], plain)
    return html.escape(escaped, quote=True).replace("`", "&#96;")


def markdown_report(report: dict[str, Any]) -> str:
    stats, source = report["statistics"], report["source"]
    lines = ["# SSH authentication forensic report", "", "Generated deterministically by CYBERY di Mariani Yuri SSH Forensic Analyzer " + VERSION + ".", "",
             "## Evidence and scope", "", f"- Source: {markdown_cell(source['filename'])}",
             f"- SHA-256: `{source['sha256']}`", f"- Source bytes: {source['bytes']}",
             f"- Timestamp interval (UTC): {markdown_cell(stats['first_timestamp'])} to {markdown_cell(stats['last_timestamp'])}",
             f"- Rolling window: {report['parameters']['window_seconds']} seconds, inclusive; threshold: {report['parameters']['threshold']} failures.", "",
             "## Parsing coverage", "", "| Category | Count |", "| --- | ---: |"]
    lines.extend(f"| {markdown_cell(key)} | {value} |" for key, value in report["coverage"].items())
    lines += ["", "## Observed records", "", "| Kind | Count |", "| --- | ---: |"]
    lines.extend(f"| {markdown_cell(key)} | {value} |" for key, value in stats["counts_by_kind"].items())
    lines += ["", "An invalid-user notice is not counted as an additional failed authentication. Repeated source records are preserved; this tool does not guess whether repeated lines describe duplicates.", "",
              "Policy notices and refused connections are contextual records, not failed authentications. A client exit code such as 255 alone does not establish that password authentication was attempted.", "",
              "## Review candidates", "", "Qualifying rolling-window endpoints may overlap. They are not counts of distinct incidents. A success after failures is not proof of compromise.", "",
              "| Rule | Endpoint (UTC) | Source | Account | Failures in window | Evidence IDs |", "| --- | --- | --- | --- | ---: | --- |"]
    for alert in report["alerts"]:
        lines.append("| " + " | ".join(markdown_cell(alert[key]) for key in ("rule", "timestamp", "source_address", "user", "failure_count")) + " | " + markdown_cell(", ".join(alert["evidence_event_ids"])) + " |")
    if not report["alerts"]:
        lines.append("| No qualifying records | — | — | — | 0 | — |")
    lines += ["", "## Threshold sensitivity", "", "| Threshold | Qualifying failure endpoints | Sources flagged | Success-after-failure candidates |", "| ---: | ---: | ---: | ---: |"]
    for item in report["threshold_sensitivity"]:
        lines.append(f"| {item['threshold']} | {item['qualifying_failure_endpoints']} | {item['sources_with_concentrated_failures']} | {item['success_after_failure_review_candidates']} |")
    lines += ["", "## Timestamp provenance", "", "| Origin | Records |", "| --- | ---: |"]
    lines.extend(f"| {markdown_cell(key)} | {value} |" for key, value in stats["timestamp_origins"].items())
    lines += ["", "## Limitations", "", "- Detection thresholds are analyst choices; this tool does not estimate real-world precision or recall.",
              "- Authentication logs do not establish the commands executed, downstream impact, user intent, or a person's identity.",
              "- Source address aggregation can conflate users behind NAT; PID and port reuse are possible.",
              "- Source precision, clock skew, log loss, rotation, buffering, and version-specific formats affect the timeline.",
              "- SHA-256 verifies byte consistency, not independent authenticity or completeness."]
    lines.extend("- " + markdown_cell(warning) for warning in report["warnings"])
    lines += ["", "See report.json for every parsed event, the original source line, and unparsed SSH or malformed JSONL records. See timeline.csv and the SVG charts for derived views.", ""]
    return "\n".join(lines)


def bar_chart(title: str, series: list[tuple[str, int]], description: str) -> str:
    """A self-contained SVG; labels are escaped, with no external scripts/fonts."""
    width, label_width, row_height = 860, 300, 40
    height = 140 + row_height * max(1, len(series))
    maximum = max((value for _, value in series), default=1) or 1
    parts = [f'<svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 {width} {height}" role="img" aria-labelledby="title desc">',
             f'<title id="title">{html.escape(title)}</title>', f'<desc id="desc">{html.escape(description)}</desc>',
             f'<rect width="{width}" height="{height}" fill="#111827"/>',
             '<g font-family="system-ui,sans-serif" fill="#e5e7eb">',
             f'<text x="24" y="38" font-size="23">{html.escape(title)}</text>',
             f'<text x="24" y="68" font-size="13">{html.escape(description)}</text>']
    if not series:
        parts.append('<text x="24" y="112" font-size="16">No observations in this view.</text>')
    for index, (label, value) in enumerate(series):
        y = 96 + index * row_height
        bar_width = value / maximum * (width - label_width - 80)
        parts += [f'<text x="24" y="{y + 21}" font-size="14">{html.escape(label)}</text>',
                  f'<rect x="{label_width}" y="{y}" width="{bar_width:.2f}" height="27" rx="4" fill="#38bdf8"/>',
                  f'<text x="{label_width + bar_width + 10:.2f}" y="{y + 21}" font-size="15">{value}</text>']
    parts += ["</g>", "</svg>", ""]
    return "\n".join(parts)


TIMELINE_FIELDS = ["event_id", "timestamp", "timestamp_origin", "original_timestamp", "collector_received_at", "kind", "method", "user", "invalid_user", "source_address", "source_port", "source_network", "penalty_state", "penalty_seconds", "policy_reason", "destination_address", "destination_port", "connection_index", "host", "daemon", "pid", "scenario", "source_line_number", "raw_message", "source_line"]


def spreadsheet_safe(value: Any) -> Any:
    # Preserve exact text in JSON; prevent active formulas when CSV is opened by a spreadsheet.
    if isinstance(value, str) and value.startswith(("=", "+", "-", "@", "\t", "\r")):
        return "'" + value
    return value


def write_outputs(report: dict[str, Any], directory: Path) -> list[Path]:
    directory.mkdir(parents=True, exist_ok=True)
    files = []
    path = directory / "report.json"
    path.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n", encoding="utf-8", newline="\n")
    files.append(path)
    path = directory / "timeline.csv"
    with path.open("w", encoding="utf-8", newline="") as stream:
        writer = csv.DictWriter(stream, fieldnames=TIMELINE_FIELDS, extrasaction="ignore")
        writer.writeheader()
        writer.writerows({key: spreadsheet_safe(event.get(key)) for key in TIMELINE_FIELDS} for event in report["events"])
    files.append(path)
    path = directory / "report.md"
    path.write_text(markdown_report(report), encoding="utf-8", newline="\n")
    files.append(path)
    stats = report["statistics"]
    charts = {"event-counts.svg": ("Observed SSH records", list(stats["counts_by_kind"].items()), "Authentication attempts and contextual records are distinct."),
              "failures-by-source.svg": ("Authentication failures by source", [(key, value.get("authentication_failure", 0)) for key, value in stats["by_source"].items() if value.get("authentication_failure", 0)], "An address identifies a network source, not a person."),
              "failures-by-hour.svg": ("Authentication failures by UTC hour", [(key, value.get("authentication_failure", 0)) for key, value in stats["by_utc_hour"].items()], "Bins reflect recorded timestamps; empty hours are omitted.")}
    charts["failures-by-minute.svg"] = ("Authentication failures by UTC minute", [(key, value.get("authentication_failure", 0)) for key, value in stats["by_utc_minute"].items()], "Bins reflect recorded timestamps; empty minutes are omitted.")
    for name, (title, series, description) in charts.items():
        path = directory / name
        path.write_text(bar_chart(title, series, description), encoding="utf-8", newline="\n")
        files.append(path)
    return files


def main(argv: list[str] | None = None) -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--input", required=True, type=Path, help="UTF-8 raw OpenSSH syslog or JSONL evidence.")
    parser.add_argument("--output", required=True, type=Path, help="Directory for deterministic derived reports.")
    parser.add_argument("--year", type=int, help="Explicit attributed year, required for BSD timestamps.")
    parser.add_argument("--timezone", help="Explicit UTC or fixed offset, required for BSD timestamps.")
    parser.add_argument("--threshold", type=int, default=5, help="Failure threshold (default 5).")
    parser.add_argument("--window", type=int, default=60, help="Inclusive rolling window in seconds (default 60).")
    parser.add_argument("--version", action="version", version=VERSION)
    args = parser.parse_args(argv)
    try:
        if args.input.resolve().parent == args.output.resolve() and args.input.name in {"report.json", "timeline.csv", "report.md", "event-counts.svg", "failures-by-source.svg", "failures-by-hour.svg", "failures-by-minute.svg"}:
            raise InputError("The output directory would overwrite the source evidence. Choose another directory.")
        parsed = read_source(args.input, args.year, args.timezone)
        report = summarize(parsed, args.threshold, args.window)
        files = write_outputs(report, args.output)
    except (InputError, OSError) as exc:
        print(f"Error: {exc}", file=sys.stderr)
        return 2
    print(json.dumps({"source_sha256": report["source"]["sha256"], "statistics": report["statistics"],
                      "coverage": report["coverage"], "review_candidate_endpoints": len(report["alerts"]),
                      "output_files": [str(path) for path in files]}, indent=2, ensure_ascii=False))
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
