#!/usr/bin/env python3
"""
TACTIK — Seal canonicalization  `deep-stable-sort-nfc-utf8-v2`
==============================================================

Standalone, dependency-free reference implementation of the exact
canonicalization the TACTIK engine applies to a seal preimage before
hashing it with SHA-256.

Algorithm (v2):
  1. Deep, recursive key sort  — every nested object, by Unicode code point
     of the NFC-normalized key (stable; duplicate keys are impossible in JSON).
  2. NFC normalization        — applied to every string leaf AND every key.
  3. No whitespace            — no spaces, no newlines, no indentation.
  4. UTF-8 encoding           — the canonical string is encoded as UTF-8
     before being fed to SHA-256.
  5. null/undefined           — serialized as the literal `null`.
  6. Arrays                   — order is PRESERVED (arrays are never sorted).

Usage
-----
  # canonicalize + hash a preimage source document
  python3 tactik-canonicalize-v2.py payload.json

  # verify an already-published canonical preimage reproduces its hash
  python3 tactik-canonicalize-v2.py --verify preimage.txt \
      fff78075c3de35169d87e8ae19955e1c62285de36220c99bdd8d69b529bcdf79

  # read from stdin
  cat payload.json | python3 tactik-canonicalize-v2.py -

Exit code 0 = match / success, 1 = mismatch or error.

Reference implementations in the engine:
  src/lib/sealCertificate.ts
  supabase/functions/_shared/seal-certificate.ts
"""

import hashlib
import json
import sys
import unicodedata

CANONICALIZATION = "deep-stable-sort-nfc-utf8-v2"


# ── JSON string escaping, byte-identical to ECMAScript JSON.stringify ────────
_ESCAPES = {
    '"': '\\"',
    "\\": "\\\\",
    "\b": "\\b",
    "\f": "\\f",
    "\n": "\\n",
    "\r": "\\r",
    "\t": "\\t",
}


def _quote(s: str) -> str:
    """JSON.stringify-compatible quoting of an already NFC-normalized string."""
    out = ['"']
    for ch in s:
        esc = _ESCAPES.get(ch)
        if esc is not None:
            out.append(esc)
        elif ch < "\x20":
            out.append("\\u%04x" % ord(ch))
        else:
            out.append(ch)
    out.append('"')
    return "".join(out)


def _number(n) -> str:
    """ECMAScript number formatting (integral floats lose the .0 suffix)."""
    if isinstance(n, bool):  # bool is a subclass of int — handled by caller
        return "true" if n else "false"
    if isinstance(n, int):
        return str(n)
    if n != n or n in (float("inf"), float("-inf")):  # NaN / Infinity
        return "null"
    if n == int(n) and abs(n) < 1e21:
        return str(int(n))
    return repr(n)


def canonicalize(value) -> str:
    """Return the canonical v2 string for a parsed JSON value."""
    if value is None:
        return "null"
    if isinstance(value, bool):
        return "true" if value else "false"
    if isinstance(value, str):
        return _quote(unicodedata.normalize("NFC", value))
    if isinstance(value, (int, float)):
        return _number(value)
    if isinstance(value, (list, tuple)):
        return "[" + ",".join(canonicalize(v) for v in value) + "]"
    if isinstance(value, dict):
        items = [(unicodedata.normalize("NFC", str(k)), v) for k, v in value.items()]
        items.sort(key=lambda kv: kv[0])
        return "{" + ",".join(_quote(k) + ":" + canonicalize(v) for k, v in items) + "}"
    raise TypeError("unsupported JSON type: %r" % type(value))


def sha256_hex(canonical: str) -> str:
    return hashlib.sha256(canonical.encode("utf-8")).hexdigest()


# ── CLI ──────────────────────────────────────────────────────────────────────
def _read(path: str) -> str:
    if path == "-":
        return sys.stdin.read()
    with open(path, "r", encoding="utf-8") as fh:
        return fh.read()


def main(argv) -> int:
    if not argv or argv[0] in ("-h", "--help"):
        print(__doc__)
        return 0

    if argv[0] == "--verify":
        if len(argv) < 3:
            print("usage: --verify <preimage-file> <expected-sha256>", file=sys.stderr)
            return 1
        raw = _read(argv[1])
        expected = argv[2].strip().lower()

        # A published preimage is ALREADY canonical. Re-canonicalizing it must
        # be a no-op — that idempotence is itself part of the verification.
        parsed = json.loads(raw)
        recanonical = canonicalize(parsed)
        idempotent = recanonical == raw.strip()

        got = sha256_hex(raw.strip())
        got_recanon = sha256_hex(recanonical)

        print("canonicalization : " + CANONICALIZATION)
        print("preimage bytes   : %d" % len(raw.strip().encode("utf-8")))
        print("idempotent       : %s" % idempotent)
        print("sha256(preimage) : " + got)
        print("sha256(recanon)  : " + got_recanon)
        print("expected         : " + expected)
        ok = got == expected and got_recanon == expected
        print("RESULT           : " + ("MATCH" if ok else "MISMATCH"))
        return 0 if ok else 1

    parsed = json.loads(_read(argv[0]))
    canonical = canonicalize(parsed)
    print(canonical)
    print("", file=sys.stderr)
    print("canonicalization : " + CANONICALIZATION, file=sys.stderr)
    print("sha256           : " + sha256_hex(canonical), file=sys.stderr)
    return 0


if __name__ == "__main__":
    sys.exit(main(sys.argv[1:]))
