#!/usr/bin/env python3
"""Build a normalized event timeline from authorized log exports.

Supports common ISO-8601, syslog, and Apache/Nginx access-log timestamps. The
output CSV can be imported into a case notebook or SIEM for reconstruction.
"""

from __future__ import annotations

import argparse
import csv
import re
from datetime import datetime, timezone
from pathlib import Path


PATTERNS = [
    ("iso8601", re.compile(r"(?P<ts>\d{4}-\d{2}-\d{2}[T ][\d:.]+(?:Z|[+-]\d{2}:?\d{2})?)")),
    ("syslog", re.compile(r"(?P<ts>[A-Z][a-z]{2}\s+\d{1,2}\s+\d{2}:\d{2}:\d{2})")),
    ("web", re.compile(r"\[(?P<ts>\d{2}/[A-Z][a-z]{2}/\d{4}:\d{2}:\d{2}:\d{2}\s+[+-]\d{4})\]")),
]


def parse_timestamp(text: str, year_hint: int) -> tuple[str, str] | None:
    for source, pattern in PATTERNS:
        match = pattern.search(text)
        if not match:
            continue
        raw = match.group("ts")
        try:
            if source == "iso8601":
                normalized = raw.replace("Z", "+00:00")
                dt = datetime.fromisoformat(normalized)
                if dt.tzinfo is None:
                    dt = dt.replace(tzinfo=timezone.utc)
            elif source == "syslog":
                dt = datetime.strptime(f"{year_hint} {raw}", "%Y %b %d %H:%M:%S").replace(tzinfo=timezone.utc)
            else:
                dt = datetime.strptime(raw, "%d/%b/%Y:%H:%M:%S %z")
            return dt.astimezone(timezone.utc).isoformat(), source
        except ValueError:
            continue
    return None


def main() -> int:
    parser = argparse.ArgumentParser(description="Normalize logs into a timeline CSV.")
    parser.add_argument("logs", nargs="+", type=Path)
    parser.add_argument("--output", type=Path, default=Path("timeline.csv"))
    parser.add_argument("--year-hint", type=int, default=datetime.now(timezone.utc).year)
    args = parser.parse_args()

    rows: list[dict[str, str | int]] = []
    for log_path in args.logs:
        with log_path.open("r", encoding="utf-8", errors="replace") as handle:
            for line_number, line in enumerate(handle, start=1):
                parsed = parse_timestamp(line, args.year_hint)
                if parsed:
                    timestamp, parser_name = parsed
                    rows.append({
                        "timestamp_utc": timestamp,
                        "source_file": str(log_path),
                        "line_number": line_number,
                        "parser": parser_name,
                        "event": line.strip(),
                    })

    rows.sort(key=lambda row: str(row["timestamp_utc"]))
    with args.output.open("w", newline="", encoding="utf-8") as handle:
        writer = csv.DictWriter(handle, fieldnames=["timestamp_utc", "source_file", "line_number", "parser", "event"])
        writer.writeheader()
        writer.writerows(rows)
    print(f"Wrote {len(rows)} timeline events to {args.output}")
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
