#!/usr/bin/env python3
"""Offline, fail-closed receiver for E01 / CVP2 RGB24-Exact.

Implementation derived from the accompanying receiver-instructions.txt.
Python 3.10+, standard library only. No network, external dictionary, or prior
receiver is used. This is the ONE canonical intact frame recipe, NOT a general
CVP2 reader. It verifies RS syndromes, but does not repair errors. Recovered
metadata and message content are never used as paths, code, or instructions.

Run: python decode_cvp2.py source/e01-cvp2-rgb24.png --out results
The output directory must be empty; failed attempts are kept, not overwritten.
"""
from __future__ import annotations

import argparse
from collections import Counter
import hashlib
import json
import math
from pathlib import Path
import re
import struct
import sys
from typing import Any
from urllib.parse import urlsplit
import zlib

PNG_HASH = '9a9d2abd38eaf45e184dccfaeabe2c0154ed48886b8f9e2e9c60d1e45a8233d5'
PNG_BYTES = 125718
OBJECT_HASH = '47f19645d26ec974b89dab7ecb920b981477655e01bc104ecd98aea15da91ae5'
OBJECT_BYTES = 100551
CONTEXT_HEAD = '1cb4608b9ae8455b7506cee71590f16b7384f4b99284fbc4975c329b0592a8cb'
REGISTRY = 'https://www.unifiedstate.us/carrier/'
LEXICON = {
    'namespace': 'https://www.unifiedstate.us/carrier/lexicon/en-moby',
    'version': '1.0.0',
    'formsSHA256': 'bdf95e4ecf569093a6ba6e9e21f089c2c8c8f5fa484757335dff2a2ea5731d27',
    'manifestSHA256': '367cccc4843c185bbec79d630d0157b396931472e99fb4f65e2aa83b9e52d3f3',
}
WIDTH = HEIGHT = 896
M, S, TILE = 200, 224, 4
K = M * M // 85
MAX_PNG_BYTES = 80 * 1024 * 1024
MAX_IDAT_BYTES = 64 * 1024 * 1024
MAX_DEPTH = 128  # additional parser resource bound, recorded in the report
HEX64 = re.compile(r'[0-9a-f]{64}\Z')
PNG_SIGNATURE = b'\x89PNG\r\n\x1a\n'
JSON_NUMBER = re.compile(rb'-?(?:0|[1-9][0-9]*)(?:\.[0-9]+)?(?:[eE][+-]?[0-9]+)?')


class Reject(ValueError):
    """A failed verification; no damaged or ambiguous data are repaired."""


def need(condition: bool, message: str) -> None:
    if not condition:
        raise Reject(message)


def sha(data: bytes) -> str:
    return hashlib.sha256(data).hexdigest()


def is_integer(value: Any) -> bool:
    return type(value) is int  # JSON booleans are not integer IDs


def unique_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
    result: dict[str, Any] = {}
    for key, value in pairs:
        need(key not in result, f'duplicate JSON key: {key!r}')
        result[key] = value
    return result


def reject_constant(value: str) -> None:
    raise Reject(f'non-JSON constant: {value}')


def strict_json(data: bytes) -> Any:
    try:
        return json.loads(data.decode('utf-8', errors='strict'),
                          object_pairs_hook=unique_object,
                          parse_constant=reject_constant)
    except (UnicodeError, json.JSONDecodeError, RecursionError) as exc:
        raise Reject(f'invalid UTF-8 JSON: {exc}') from exc


def canonical_json(value: Any) -> bytes:
    r"""JSON.stringify-compatible for this profile's ordered objects/integers.

    Python retains insertion order. Metadata numbers have already been checked
    to be safe integers. Normal Unicode remains literal; lone surrogate code
    points (if any) receive lowercase \uXXXX escapes, as in JSON.stringify.
    """
    text = json.dumps(value, ensure_ascii=False, allow_nan=False,
                      separators=(',', ':'))
    text = ''.join('\\u%04x' % ord(c) if 0xD800 <= ord(c) <= 0xDFFF else c
                   for c in text)
    return text.encode('utf-8')


def pretty_json(value: Any) -> bytes:
    text = json.dumps(value, ensure_ascii=False, allow_nan=False, indent=2)
    text = ''.join('\\u%04x' % ord(c) if 0xD800 <= ord(c) <= 0xDFFF else c
                   for c in text)
    return (text + '\n').encode('utf-8')


def keys_are(value: Any, keys: list[str], label: str) -> None:
    need(isinstance(value, dict) and list(value) == keys,
         f'{label}: wrong object keys/order')


def hash_string(value: Any) -> bool:
    return isinstance(value, str) and HEX64.fullmatch(value) is not None


def absolute_http_url(value: Any) -> bool:
    if not isinstance(value, str):
        return False
    try:
        parsed = urlsplit(value)
        return parsed.scheme in ('http', 'https') and bool(parsed.netloc) and bool(parsed.hostname)
    except ValueError:
        return False


class Evidence:
    def __init__(self) -> None:
        self.report: dict[str, Any] = {
            'receiver': 'e01-cvp2-independent-python/1',
            'status': 'in-progress',
            'profile': 'CVP2 RGB24-Exact 1.0.0 / one intact, canonical, uncompressed frame',
            'checks': [],
            'diagnostics': {},
            'additional_checks_or_bounds': [
                'UTF-8 source-span JSON parser has a nesting resource limit of 128.',
                'A separate polynomial-division encoder reproduces all RS codewords.',
                'USL1 metadata and wire tokens are re-encoded for byte-for-byte equality.',
                'A separate source-span JSON parser is cross-checked against Python JSON parsing.',
            ],
            'capabilities_not_claimed': [
                'Reed-Solomon error repair', 'rotation/perspective/colour correction',
                'resized, nonuniform, JPEG, or optical-channel decoding',
                'CVP2 gzip, multi-frame sequences, or arbitrary grid sizes',
                'external authentication, online status, or truth of content',
                'lexical dictionary verification (zero lexical tokens)',
            ],
        }
        self.current = ''

    def stage(self, label: str, function: Any, *args: Any) -> Any:
        self.current = label
        try:
            value, details = function(*args)
        except Exception as exc:
            self.report['checks'].append({
                'stage': label, 'status': 'FAIL',
                'error_type': type(exc).__name__, 'error': str(exc),
            })
            raise
        self.report['checks'].append({'stage': label, 'status': 'PASS', **details})
        return value


def check_artifact(blob: bytes) -> tuple[bytes, dict[str, Any]]:
    need(len(blob) <= MAX_PNG_BYTES, 'PNG exceeds 80 MiB limit')
    need(len(blob) == PNG_BYTES, 'unexpected original PNG byte count')
    actual = sha(blob)
    need(actual == PNG_HASH, 'original PNG SHA-256 mismatch')
    return blob, {'bytes': len(blob), 'sha256': actual}


def png_container(blob: bytes) -> tuple[bytes, dict[str, Any]]:
    need(len(blob) <= MAX_PNG_BYTES, 'PNG exceeds 80 MiB limit')
    need(blob[:8] == PNG_SIGNATURE, 'PNG signature mismatch')
    pos = 8
    chunks: list[dict[str, Any]] = []
    idats: list[bytes] = []
    idat_total = 0
    seen_ihdr = seen_srgb = seen_iend = False
    while pos < len(blob):
        start = pos
        need(pos + 12 <= len(blob), 'truncated PNG chunk framing')
        length = struct.unpack_from('>I', blob, pos)[0]
        typ = blob[pos + 4:pos + 8]
        need(length <= len(blob) - pos - 12, 'PNG chunk length reads beyond file')
        data = blob[pos + 8:pos + 8 + length]
        supplied = struct.unpack_from('>I', blob, pos + 8 + length)[0]
        calculated = zlib.crc32(typ + data) & 0xFFFFFFFF
        need(supplied == calculated, f'PNG {typ!r} chunk CRC mismatch at byte {pos}')
        pos += length + 12
        need(typ in (b'IHDR', b'sRGB', b'IDAT', b'IEND'),
             f'forbidden PNG chunk: {typ!r}')
        if not seen_ihdr:
            need(typ == b'IHDR', 'IHDR must be the first PNG chunk')
        if typ == b'IHDR':
            need(not seen_ihdr and not chunks and length == 13,
                 'IHDR duplicated, misplaced, or wrong length')
            values = struct.unpack('>IIBBBBB', data)
            need(values == (896, 896, 8, 2, 0, 0, 0), f'unsupported IHDR: {values}')
            seen_ihdr = True
        elif typ == b'sRGB':
            need(seen_ihdr and not seen_srgb and not idats,
                 'sRGB duplicated or misplaced')
            need(data == b'\0', 'sRGB must have one zero intent byte')
            seen_srgb = True
        elif typ == b'IDAT':
            need(seen_ihdr and not seen_iend, 'IDAT misplaced')
            # The only allowed chunks before IEND are IHDR, optional sRGB, IDAT;
            # sRGB is forbidden after the first IDAT, so IDATs are consecutive.
            idats.append(data)
            idat_total += length
            need(len(idats) <= 4096, 'more than 4096 IDAT chunks')
            need(idat_total <= MAX_IDAT_BYTES, 'combined IDAT exceeds 64 MiB')
        elif typ == b'IEND':
            need(seen_ihdr and bool(idats) and not seen_iend and length == 0,
                 'IEND misplaced/duplicated/nonempty, or no IDAT')
            seen_iend = True
            need(pos == len(blob), 'bytes or chunks after IEND')
        chunks.append({'type': typ.decode('ascii'), 'offset': start,
                       'data_bytes': length, 'crc32': f'{supplied:08x}',
                       'crc_verified': True})
    need(seen_ihdr and seen_iend and bool(idats), 'incomplete PNG container')
    return b''.join(idats), {
        'width': WIDTH, 'height': HEIGHT, 'bit_depth': 8, 'colour_type': 'RGB (2)',
        'compression': 0, 'filter_method': 0, 'interlace': 0,
        'chunks': chunks, 'every_chunk_crc_verified': True,
        'idat_chunks': len(idats), 'idat_bytes': idat_total,
        'sRGB_intent_zero_present': seen_srgb, 'trailing_bytes': 0,
    }


def paeth(a: int, b: int, c: int) -> int:
    p = a + b - c
    pa, pb, pc = abs(p - a), abs(p - b), abs(p - c)
    if pa <= pb and pa <= pc:
        return a
    return b if pb <= pc else c


def unfilter_rows(expanded: bytes, width: int, height: int) -> tuple[bytes, dict[int, int]]:
    stride = width * 3
    need(len(expanded) == height * (stride + 1), 'wrong inflated scanline byte count')
    raster = bytearray(height * stride)
    counts: Counter[int] = Counter()
    previous = bytearray(stride)
    for y in range(height):
        start = y * (stride + 1)
        f = expanded[start]
        need(f in range(5), f'invalid PNG scanline filter {f} on row {y}')
        counts[f] += 1
        row = bytearray(expanded[start + 1:start + 1 + stride])
        if f != 0:
            for i in range(stride):
                left = row[i - 3] if i >= 3 else 0
                above = previous[i]
                upper_left = previous[i - 3] if i >= 3 else 0
                if f == 1:
                    predictor = left
                elif f == 2:
                    predictor = above
                elif f == 3:
                    predictor = (left + above) // 2
                else:
                    predictor = paeth(left, above, upper_left)
                row[i] = (row[i] + predictor) & 255
        raster[y * stride:(y + 1) * stride] = row
        previous = row
    return bytes(raster), dict(counts)


def inflate_rgb(idat: bytes) -> tuple[bytes, dict[str, Any]]:
    expected = HEIGHT * (WIDTH * 3 + 1)
    need(expected == 2409344, 'internal raster-size calculation error')
    inflater = zlib.decompressobj()
    try:
        raw = inflater.decompress(idat, expected + 1)
    except zlib.error as exc:
        raise Reject(f'invalid zlib stream: {exc}') from exc
    need(len(raw) <= expected, 'zlib output exceeds bound')
    need(inflater.eof, 'truncated zlib stream')
    need(not inflater.unconsumed_tail, 'unconsumed zlib input')
    need(not inflater.unused_data, 'trailing compressed data or additional zlib stream')
    need(len(raw) == expected, 'wrong decompressed size')
    rgb, filters = unfilter_rows(raw, WIDTH, HEIGHT)
    return rgb, {
        'zlib_streams': 1, 'inflated_bytes': len(raw), 'rgb_sample_bytes': len(rgb),
        'row_filters': filters, 'raw_rgb_sha256': sha(rgb),
        'decoder': 'local standard-library zlib plus explicit PNG inverse filters',
        'display_colour_transforms_applied': False,
    }


def exact_modules(rgb: bytes) -> tuple[bytes, dict[str, Any]]:
    need(len(rgb) == WIDTH * HEIGHT * 3, 'RGB raster length mismatch')
    modules = bytearray(S * S * 3)
    for my in range(S):
        for mx in range(S):
            pos = ((my * TILE) * WIDTH + mx * TILE) * 3
            sample = rgb[pos:pos + 3]
            expected_row = sample * TILE
            for dy in range(TILE):
                p = pos + dy * WIDTH * 3
                need(rgb[p:p + TILE * 3] == expected_row,
                     f'nonuniform 4x4 module at ({mx},{my}), physical row offset {dy}')
            idx = (my * S + mx) * 3
            modules[idx:idx + 3] = sample
    return bytes(modules), {
        'logical_modules': S * S, 'physical_pixels_checked': WIDTH * HEIGHT,
        'physical_pixels_per_module': 16, 'nonuniform_modules': 0,
        'logical_rgb_sha256': sha(bytes(modules)),
    }


def expected_frame() -> bytes:
    frame = bytearray(b'\xff\xff\xff' * (S * S))
    def put(x: int, y: int, colour: tuple[int, int, int]) -> None:
        p = (y * S + x) * 3
        frame[p:p + 3] = bytes(colour)
    colours = [(255, 0, 0), (0, 255, 0), (0, 0, 255), (255, 255, 0)]
    starts = [(2, 2), (S - 9, 2), (S - 9, S - 9), (2, S - 9)]
    for (x, y), colour in zip(starts, colours):
        for dy in range(7):
            for dx in range(7):
                c = ((0, 0, 0) if dx in (0, 6) or dy in (0, 6)
                     else colour if 2 <= dx <= 4 and 2 <= dy <= 4
                     else (255, 255, 255))
                put(x + dx, y + dy, c)
    for y in (3, S - 5):
        x = (S - 32) // 2
        for v in range(16):
            colour = (85 * (v >> 2), 255 * ((v >> 1) & 1), 255 * (v & 1))
            for dy in range(2):
                for dx in range(2):
                    put(x + 2 * v + dx, y + dy, colour)
    for p in range(12, 12 + M):
        c = (0, 0, 0) if p % 2 == 0 else (255, 255, 255)
        for x, y in ((10, p), (S - 11, p), (p, 10), (p, S - 11)):
            put(x, y, c)
    return bytes(frame)


def validate_frame(modules: bytes) -> tuple[bytes, dict[str, Any]]:
    need(len(modules) == S * S * 3, 'module raster length mismatch')
    frame = expected_frame()
    compared = 0
    for y in range(S):
        for x in range(S):
            if 12 <= x < 12 + M and 12 <= y < 12 + M:
                continue
            p = (y * S + x) * 3
            need(modules[p:p + 3] == frame[p:p + 3],
                 f'non-data module mismatch at ({x},{y})')
            compared += 1
    data = b''.join(modules[(y * S + 12) * 3:(y * S + 12 + M) * 3]
                    for y in range(12, 12 + M))
    need(len(data) == 3 * M * M, 'data-square extraction length mismatch')
    used = 255 * K
    need(K == 470 and used == 119850, 'incorrect codeword calculation')
    need(not any(data[used:]), 'nonzero unused RGB data modules')
    return data[:used], {
        'non_data_modules_compared': compared, 'non_data_mismatches': 0,
        'data_modules': M * M, 'bytes_per_data_module': 3,
        'used_data_modules': used // 3, 'unused_black_modules': (len(data) - used) // 3,
        'wire_bytes': used, 'wire_sha256': sha(data[:used]),
        'distinct_data_rgb_values': len({data[i:i+3] for i in range(0, len(data), 3)}),
        'reference_strips_used_for_colour_correction': False,
    }


def gf_mul(a: int, b: int) -> int:
    result = 0
    while b:
        if b & 1:
            result ^= a
        b >>= 1
        a <<= 1
        if a & 0x100:
            a ^= 0x11D
    return result


def roots() -> list[int]:
    result, value = [], 1
    for _ in range(32):
        result.append(value)
        value = gf_mul(value, 2)
    return result


def rs_verify(wire: bytes) -> tuple[list[bytes], dict[str, Any]]:
    need(len(wire) == 255 * K, 'wrong RS-interleaved wire length')
    words = [wire[k::K] for k in range(K)]
    tables = [[gf_mul(a, root) for a in range(256)] for root in roots()]
    nonzero = []
    for k, word in enumerate(words):
        need(len(word) == 255, 'wrong RS codeword length')
        for j, table in enumerate(tables):
            s = 0
            for b in word:
                s = table[s] ^ b
            if s:
                nonzero.append({'codeword': k, 'root_exponent': j, 'syndrome': s})
    need(not nonzero, f'{len(nonzero)} nonzero RS syndromes; first failures: {nonzero[:8]}')
    return words, {
        'codewords': K, 'codeword_bytes': 255, 'data_bytes_per_word': 223,
        'parity_bytes_per_word': 32, 'syndromes_checked': K * 32,
        'nonzero_syndromes': 0, 'primitive_polynomial': '0x11D',
        'primitive_element': 2, 'repairs_attempted': 0,
    }


def rs_reencode_diagnostic(words: list[bytes]) -> tuple[None, dict[str, Any]]:
    generator = [1]
    for root in roots():
        new = [0] * (len(generator) + 1)
        for i, coefficient in enumerate(generator):
            new[i] ^= coefficient
            new[i + 1] ^= gf_mul(coefficient, root)
        generator = new
    tables = [[gf_mul(coefficient, v) for v in range(256)]
              for coefficient in generator[1:]]
    for k, word in enumerate(words):
        work = bytearray(word[:223] + bytes(32))
        for i in range(223):
            coefficient = work[i]
            if coefficient:
                for j, table in enumerate(tables, start=1):
                    work[i + j] ^= table[coefficient]
        need(word[:223] + bytes(work[-32:]) == word, f'RS re-encoding diagnostic mismatch: {k}')
    return None, {'additional_diagnostic': True, 'reencoded_codewords': len(words),
                  'byte_identical_codewords': len(words)}


def cvp2_object(words: list[bytes]) -> tuple[tuple[bytes, bytes], dict[str, Any]]:
    need(len(words) == K and all(len(w) == 255 for w in words), 'invalid codeword shape')
    h = words[0][:223]
    def u16(offset: int) -> int:
        return struct.unpack_from('<H', h, offset)[0]
    def u32(offset: int) -> int:
        return struct.unpack_from('<I', h, offset)[0]
    need(zlib.crc32(b'123456789') == 0xCBF43926, 'CRC implementation check failed')
    need(h[:4] == b'UCV2', 'wrong CVP2 wire magic')
    need(h[4:8] == bytes([2, 0, 2, 1]), 'wrong CVP2 version/flags/RGB/ECC profiles')
    need((u16(8), u16(10)) == (223, M), 'wrong CVP2 header or grid length')
    need((u32(12), u32(16)) == (0, 1), 'not the expected single frame')
    L = u32(20)
    need((L, u32(24), u32(28), u32(32)) == (OBJECT_BYTES, OBJECT_BYTES, OBJECT_BYTES, 0),
         'wrong CVP2 lengths or stream offset')
    crc = u32(219)
    need(crc == zlib.crc32(h[:219]), 'CVP2 header CRC mismatch')
    q = u16(148)
    need(1 <= q <= 69, 'invalid CVP2 metadata length')
    raw_meta = h[150:150 + q]
    need(not any(h[150 + q:219]), 'nonzero CVP2 metadata padding')
    metadata = strict_json(raw_meta)
    keys_are(metadata, ['n', 't'], 'CVP2 metadata')
    need(all(isinstance(metadata[k], str) for k in ('n', 't')), 'non-string CVP2 metadata')
    name = metadata['n']
    need(bool(name) and all(ord(c) >= 32 and c not in '\\/<>:"|?*' for c in name),
         'unsafe descriptive filename')
    need(metadata == {'n': 'e01.usl', 't': 'application/octet-stream'},
         'unexpected CVP2 metadata values')
    need(canonical_json(metadata) == raw_meta, 'noncanonical CVP2 metadata')
    available = b''.join(w[:223] for w in words[1:])
    need(len(available) == 104587 and L <= len(available), 'CVP2 payload capacity mismatch')
    obj = available[:L]
    need(not any(available[L:]), 'nonzero CVP2 payload padding')
    digest = hashlib.sha256(obj).digest()
    for label, offset in [('chunk', 36), ('stream', 68), ('original', 100)]:
        need(h[offset:offset + 32] == digest, f'CVP2 {label} SHA-256 mismatch')
    need(h[132:148] == digest[:16], 'CVP2 message-ID prefix mismatch')
    need(digest.hex() == OBJECT_HASH, 'recovered USL1 does not match expected SHA-256')
    return (obj, h), {
        'magic': 'UCV2', 'wire_version': 2, 'flags': 0, 'rgb_profile': 2, 'ecc_profile': 1,
        'header_bytes': 223, 'grid': M, 'frame_index': 0, 'frame_count': 1,
        'header_crc32': f'{crc:08x}', 'metadata_bytes': q, 'metadata': metadata,
        'metadata_zero_padding_bytes': 69 - q, 'payload_capacity_bytes': len(available),
        'object_bytes': L, 'payload_zero_padding_bytes': len(available) - L,
        'chunk_sha256': digest.hex(), 'stream_sha256': digest.hex(),
        'original_sha256': digest.hex(), 'message_id': digest[:16].hex(),
        'all_hash_fields_and_expected_object_hash_match': True,
    }


def uleb_encode(value: int) -> bytes:
    need(is_integer(value) and 0 <= value <= 0xFFFFFFFF, 'ULEB128 encode value out of range')
    result = bytearray()
    while True:
        group = value & 127
        value >>= 7
        result.append(group | (128 if value else 0))
        if not value:
            return bytes(result)


def uleb_read(data: bytes, pos: int) -> tuple[int, int]:
    start, value = pos, 0
    for i in range(5):
        need(pos < len(data), 'truncated ULEB128')
        b = data[pos]
        pos += 1
        value |= (b & 127) << (7 * i)
        need(value <= 0xFFFFFFFF, 'ULEB128 overflow')
        if not b & 128:
            need(data[start:pos] == uleb_encode(value), 'nonminimal ULEB128')
            return value, pos
    raise Reject('ULEB128 exceeds five bytes')


def validate_usl_metadata(raw: bytes) -> dict[str, Any]:
    metadata = strict_json(raw)
    keys_are(metadata, ['format', 'lexicon', 'concepts'], 'USL1 metadata')
    need(metadata['format'] == 'usl-lexical-message/1', 'wrong USL1 application format')
    lexicon = metadata['lexicon']
    keys_are(lexicon, ['namespace', 'version', 'formsSHA256', 'manifestSHA256'], 'lexicon')
    need(lexicon == LEXICON, 'lexical release pins mismatch')
    need(hash_string(lexicon['formsSHA256']) and hash_string(lexicon['manifestSHA256']),
         'malformed lexical release hashes')
    concepts = metadata['concepts']
    need(isinstance(concepts, list), 'concept table is not a list')
    seen = set()
    for i, pin in enumerate(concepts):
        keys_are(pin, ['registry', 'tier', 'id', 'head'], f'concept pin {i}')
        need(absolute_http_url(pin['registry']), 'non-absolute HTTP(S) concept registry')
        need(pin['tier'] == 'checked', 'concept pin is not checked tier')
        need(is_integer(pin['id']) and 0 <= pin['id'] <= 9007199254740991,
             'invalid concept ID')
        need(hash_string(pin['head']), 'invalid concept head hash')
        identity = tuple(pin[k] for k in ['registry', 'tier', 'id', 'head'])
        need(identity not in seen, 'duplicate concept reference')
        seen.add(identity)
    need(canonical_json(metadata) == raw, 'noncanonical USL1 metadata bytes')
    return metadata


def unpack_usl(obj: bytes) -> tuple[dict[str, Any], dict[str, Any]]:
    need(len(obj) >= 64, 'truncated USL1 header')
    need(obj[:4] == b'USL1' and obj[4:6] == b'\x01\0', 'wrong USL1 magic/version/flags')
    header_len = struct.unpack_from('<H', obj, 6)[0]
    m, N, E, B = struct.unpack_from('<IIII', obj, 8)
    need(header_len == 64 and obj[24:32] == bytes(8), 'USL1 header/reserved fields mismatch')
    need(1 <= m <= 1048576 and len(obj) == 64 + m + B and N <= B,
         'USL1 metadata/body lengths invalid')
    need(N <= 134217728 and E <= 134217728, 'USL1 token or expansion limit exceeded')
    need((m, N, E, B) == (1970, 23, 98470, 98517), 'unexpected E01 USL1 header counts')
    internal_hash = hashlib.sha256(obj[:32] + obj[64:]).digest()
    need(internal_hash == obj[32:64], 'USL1 internal SHA-256 mismatch')
    metadata = validate_usl_metadata(obj[64:64 + m])
    body = obj[64 + m:]
    pos = expanded_length = 0
    expanded_parts, tokens, provenance, wire_records = [], [], [], []
    counts: Counter[int] = Counter()
    use_order: list[int] = []
    use_counts: Counter[int] = Counter()
    for number in range(N):
        body_start = pos
        operation, pos = uleb_read(body, pos)
        tag, p = operation % 4, operation // 4
        counts[tag] += 1
        record: dict[str, Any] = {'token_index': number, 'wire_tag': tag}
        if tag == 0:
            need(p > 0, 'zero-length literal token')
            size = p
        elif tag == 1:
            raise Reject('lexical token outside the supplied experiment profile')
        elif tag == 2:
            need(p < len(metadata['concepts']), 'unknown concept-table index')
            size, pos = uleb_read(body, pos)
            need(size > 0, 'empty concept surface')
            record['concept_table_index'] = p
            if p not in use_counts:
                use_order.append(p)
            use_counts[p] += 1
        else:
            need(p == 0, 'unknown tag-3 operation')
            size = 0
        payload_start = pos
        need(pos + size <= len(body), 'truncated USL1 token payload')
        raw = body[pos:pos + size] if tag != 3 else b' '
        pos += size
        try:
            text = raw.decode('utf-8', errors='strict')
        except UnicodeError as exc:
            raise Reject(f'token {number} is not strict UTF-8') from exc
        need(expanded_length + len(raw) <= E, 'token expansion exceeds declared byte count')
        record.update({'body_byte_span': [body_start, pos],
                       'surface_body_byte_span': [payload_start, pos],
                       'expanded_utf8_byte_span': [expanded_length, expanded_length + len(raw)],
                       'expanded_utf8_bytes': len(raw)})
        if tag == 2:
            token = {'type': 'concept', **metadata['concepts'][p], 'text': text}
            record['surface'] = text
            record['type'] = 'concept'
        else:
            token = {'type': 'literal', 'text': text}
            record['type'] = 'literal'
        tokens.append(token)
        provenance.append(record)
        wire_records.append({'tag': tag, 'parameter': p, 'raw': raw})
        expanded_parts.append(raw)
        expanded_length += len(raw)
    need(pos == B, 'unconsumed USL1 body bytes')
    need(expanded_length == E, 'expanded UTF-8 length mismatch')
    need(use_order == list(range(len(metadata['concepts']))),
         'unused references or table not in first-use order')
    need(len(metadata['concepts']) == 11 and counts == Counter({0: 12, 2: 11}),
         'unexpected E01 concept table size or token tags')
    need(all(use_counts[i] == 1 for i in range(11)), 'E01 references not used exactly once')
    expanded = b''.join(expanded_parts)
    view = {'format': metadata['format'], 'lexicon': metadata['lexicon'],
            'utf8Bytes': E, 'tokens': tokens}
    decoded = {'metadata': metadata, 'view': view, 'expanded': expanded,
               'provenance': provenance, 'wire_records': wire_records,
               'header': {'m': m, 'N': N, 'E': E, 'B': B}}
    return decoded, {
        'magic': 'USL1', 'wire_version': 1, 'flags': 0, 'header_bytes': 64,
        'metadata_bytes': m, 'body_bytes': B, 'expanded_utf8_bytes': E, 'tokens': N,
        'internal_sha256': internal_hash.hex(), 'canonical_metadata_verified': True,
        'literal_tokens': counts[0], 'lexical_tokens': counts[1], 'concept_tokens': counts[2],
        'space_tokens': counts[3], 'unique_references': len(use_counts),
        'first_use_order': use_order, 'all_references_used_exactly_once': True,
        'body_consumed_bytes': pos, 'reconstructed_source_sha256': sha(expanded),
        'lexical_dataset_downloaded_or_verified': False,
    }


def reencode_usl(obj: bytes, decoded: dict[str, Any]) -> tuple[None, dict[str, Any]]:
    body = bytearray()
    for record in decoded['wire_records']:
        tag, p, raw = record['tag'], record['parameter'], record['raw']
        if tag == 0:
            body += uleb_encode(4 * len(raw)) + raw
        elif tag == 2:
            body += uleb_encode(4 * p + 2) + uleb_encode(len(raw)) + raw
        elif tag == 3:
            body += uleb_encode(3)
        else:
            raise Reject('unsupported tag in re-encoding diagnostic')
    metadata = canonical_json(decoded['metadata'])
    h = bytearray(64)
    h[:4] = b'USL1'
    h[4] = 1
    struct.pack_into('<HIIII', h, 6, 64, len(metadata), len(decoded['wire_records']),
                     len(decoded['expanded']), len(body))
    h[32:64] = hashlib.sha256(bytes(h[:32]) + metadata + bytes(body)).digest()
    rebuilt = bytes(h) + metadata + bytes(body)
    need(rebuilt == obj, 'USL1 re-encoding diagnostic differs from extracted binary')
    return None, {'additional_diagnostic': True, 'byte_identical': True,
                  'reencoded_bytes': len(rebuilt), 'sha256': sha(rebuilt)}


def json_path(path: tuple[Any, ...]) -> str:
    result = '$'
    for item in path:
        if is_integer(item):
            result += f'[{item}]'
        elif re.fullmatch(r'[A-Za-z_][A-Za-z_0-9]*', item):
            result += '.' + item
        else:
            result += '[' + json.dumps(item, ensure_ascii=True) + ']'
    return result


class SpanJSONParser:
    """Strict JSON parser whose offsets are original UTF-8 BYTES, not characters.

    Keys are parsed separately and are never mistaken for string values. All
    decoded object keys are duplicate-checked, including escaped duplicates.
    No literal evaluation or execution is performed.
    """
    def __init__(self, source: bytes) -> None:
        try:
            source.decode('utf-8', errors='strict')
        except UnicodeError as exc:
            raise Reject('source is not UTF-8') from exc
        self.source, self.pos = source, 0
        self.strings: list[dict[str, Any]] = []
        self.max_depth_seen = 0

    def ws(self) -> None:
        while self.pos < len(self.source) and self.source[self.pos] in b' \t\r\n':
            self.pos += 1

    def char(self, b: int) -> None:
        need(self.pos < len(self.source) and self.source[self.pos] == b,
             f'expected JSON byte {b} at {self.pos}')
        self.pos += 1

    def string(self, path: tuple[Any, ...], is_key: bool = False) -> str:
        start = self.pos
        self.char(34)
        while self.pos < len(self.source):
            b = self.source[self.pos]
            if b == 34:
                self.pos += 1
                raw = self.source[start:self.pos]
                value = strict_json(raw)
                need(isinstance(value, str), 'JSON string parse error')
                if not is_key:
                    self.strings.append({'path': path, 'json_path': json_path(path),
                                         'quote_byte_span': [start, self.pos],
                                         'content_byte_span': [start + 1, self.pos - 1],
                                         'value': value,
                                         'source_contains_escape': b'\\' in raw[1:-1]})
                return value
            need(b >= 32, f'unescaped JSON control byte at {self.pos}')
            if b == 92:
                self.pos += 2  # final JSON string parse validates escape contents
            else:
                self.pos += 1
        raise Reject('unterminated JSON string')

    def value(self, path: tuple[Any, ...] = (), depth: int = 0) -> Any:
        need(depth <= MAX_DEPTH, 'JSON nesting exceeds local 128-depth resource bound')
        self.max_depth_seen = max(self.max_depth_seen, depth)
        self.ws()
        need(self.pos < len(self.source), 'unexpected end of JSON')
        first = self.source[self.pos]
        if first == 34:
            return self.string(path)
        if first == 123:
            self.pos += 1
            obj: dict[str, Any] = {}
            self.ws()
            if self.pos < len(self.source) and self.source[self.pos] == 125:
                self.pos += 1
                return obj
            while True:
                self.ws()
                key = self.string(path, is_key=True)
                need(key not in obj, f'duplicate JSON key at {json_path(path)}: {key!r}')
                self.ws()
                self.char(58)
                obj[key] = self.value(path + (key,), depth + 1)
                self.ws()
                need(self.pos < len(self.source), 'unterminated JSON object')
                if self.source[self.pos] == 125:
                    self.pos += 1
                    return obj
                self.char(44)
        if first == 91:
            self.pos += 1
            array = []
            self.ws()
            if self.pos < len(self.source) and self.source[self.pos] == 93:
                self.pos += 1
                return array
            while True:
                array.append(self.value(path + (len(array),), depth + 1))
                self.ws()
                need(self.pos < len(self.source), 'unterminated JSON array')
                if self.source[self.pos] == 93:
                    self.pos += 1
                    return array
                self.char(44)
        for literal, value in [(b'true', True), (b'false', False), (b'null', None)]:
            if self.source.startswith(literal, self.pos):
                self.pos += len(literal)
                return value
        match = JSON_NUMBER.match(self.source, self.pos)
        need(match is not None, f'invalid JSON value at byte {self.pos}')
        self.pos = match.end()
        return strict_json(match.group())

    def parse(self) -> Any:
        value = self.value()
        self.ws()
        need(self.pos == len(self.source), 'trailing bytes after reconstructed JSON')
        return value


def resolve_context(decoded: dict[str, Any]) -> tuple[dict[str, Any], dict[str, Any]]:
    source = decoded['expanded']
    parser = SpanJSONParser(source)
    document = parser.parse()
    need(document == strict_json(source), 'source-span parser differs from standard JSON parser')
    need(isinstance(document, dict) and document.get('format') == 'carrier-expression-experiment/1',
         'wrong reconstructed application format')
    for key in ['context', 'grammar', 'expression']:
        need(isinstance(document.get(key), dict), f'missing/non-object top-level {key}')
    context = document['context']
    need(context.get('registry') == REGISTRY and context.get('tier') == 'checked'
         and context.get('head') == CONTEXT_HEAD, 'context registry/tier/head mismatch')
    checked_jsonl = context.get('checkedJSONL')
    need(isinstance(checked_jsonl, str), 'checkedJSONL is not a string')
    try:
        corpus = checked_jsonl.encode('utf-8', errors='strict')
    except UnicodeError as exc:
        raise Reject('checkedJSONL cannot be encoded as strict UTF-8') from exc
    need(corpus.endswith(b'\n'), 'checkedJSONL is not LF-terminated')
    need(b'\r' not in corpus and b'\xef\xbb\xbf' not in corpus, 'CR/BOM in checkedJSONL')
    lines = corpus.split(b'\n')[:-1]
    need(len(lines) == 106 and all(line.strip() for line in lines),
         'checkedJSONL must have exactly 106 nonblank complete records')
    records = []
    ledger = []
    missing_checked = []
    head = 'genesis'
    for i, line in enumerate(lines):
        record = strict_json(line)
        need(isinstance(record, dict), f'context record {i} is not an object')
        need(is_integer(record.get('id')) and record['id'] == i,
             f'context record {i} has invalid/sequential-ID mismatch')
        need(record.get('prev') == head, f'context record {i} prev mismatch')
        need(record.get('checked') is not False, f'context record {i} contradicts checked tier')
        if 'checked' not in record:
            missing_checked.append(i)
        need(isinstance(record.get('name'), str) and isinstance(record.get('definition'), str),
             f'context record {i} name/definition not strings')
        prev = head
        head = sha(line)
        ledger.append({'id': i, 'line_bytes_without_LF': len(line), 'prev': prev, 'sha256': head})
        records.append(record)
    need(head == context['head'], 'final checkedJSONL chain head mismatch')
    relation_count = 0
    for record in records:
        relations = record.get('relations', [])
        need(isinstance(relations, list), f'relations not a list for context ID {record["id"]}')
        for identifier in relations:
            need(is_integer(identifier) and 0 <= identifier < len(records),
                 f'unresolved relation ID in context entry {record["id"]}')
            relation_count += 1
    for pin in decoded['metadata']['concepts']:
        need(pin['registry'] == context['registry'] and pin['tier'] == context['tier']
             and pin['head'] == context['head'], 'selected concept context pin mismatch')
        need(pin['id'] < len(records), 'selected concept ID absent from context')
    resolved = {'document': document, 'records': records, 'corpus': corpus,
                'chain': ledger, 'strings': parser.strings, 'parser_depth': parser.max_depth_seen}
    return resolved, {
        'format': document['format'], 'registry': context['registry'], 'tier': context['tier'],
        'records': len(records), 'all_ids_sequential': True,
        'all_prev_links_verified': True, 'final_head': head,
        'checked_jsonl_bytes': len(corpus), 'whole_checked_jsonl_sha256': sha(corpus),
        'resolved_relation_ids': relation_count, 'selected_pins_verified': len(decoded['metadata']['concepts']),
        'legacy_records_without_checked_property': missing_checked,
        'legacy_absence_permitted_by_recipe': True, 'explicit_checked_false_records': [],
        'source_span_parser_maximum_nesting': parser.max_depth_seen,
        'source_span_parser_agrees_with_standard_JSON': True,
    }


def clause_concept_paths(value: Any, path: tuple[Any, ...]) -> list[tuple[Any, ...]]:
    paths = []
    if isinstance(value, dict):
        if 'concept' in value:
            need(isinstance(value['concept'], str), f'non-string clause concept at {json_path(path)}')
            paths.append(path + ('concept',))
        for key, item in value.items():
            paths.extend(clause_concept_paths(item, path + (key,)))
    elif isinstance(value, list):
        for i, item in enumerate(value):
            paths.extend(clause_concept_paths(item, path + (i,)))
    return paths


def node_at(value: Any, path: tuple[Any, ...]) -> Any:
    for item in path:
        value = value[item]
    return value


def bind_concepts(decoded: dict[str, Any], resolved: dict[str, Any]) -> tuple[list[dict[str, Any]], dict[str, Any]]:
    source = decoded['expanded']
    document = resolved['document']
    expression = document['expression']
    need(isinstance(expression.get('clauses'), list), 'expression clauses must be a list')
    required_paths = clause_concept_paths(expression['clauses'], ('expression', 'clauses'))
    need(len(required_paths) == 11 and len(set(required_paths)) == 11,
         'expected eleven unique concept-valued clause properties')
    values_by_span = {tuple(s['content_byte_span']): s for s in resolved['strings']}
    consumed = set()
    bindings = []
    for token in decoded['provenance']:
        if token['wire_tag'] != 2:
            continue
        index = token['concept_table_index']
        surface = token['surface']
        need(surface == f'C{index}', 'concept alias differs from actual table index')
        start, end = token['expanded_utf8_byte_span']
        need(0 < start <= end < len(source), 'concept span out of source bounds')
        need(source[start - 1:start] == b'"' and source[end:end + 1] == b'"',
             'concept token is not exactly surrounded by JSON string quotes')
        need(source[start:end] == surface.encode('utf-8'), 'concept byte-span/surface mismatch')
        item = values_by_span.get((start, end))
        need(item is not None, 'concept token does not occupy one complete JSON string value')
        path = item['path']
        need(not item['source_contains_escape'] and item['value'] == surface,
             'concept surface is escaped or decoded value mismatch')
        need(path in required_paths and path not in consumed,
             'concept token outside expression clause or duplicate binding')
        consumed.add(path)
        pin = decoded['metadata']['concepts'][index]
        record = resolved['records'][pin['id']]
        clause = node_at(document, path[:-1])
        bindings.append({
            'alias': surface, 'table_index': index, 'token_index': token['token_index'],
            'expanded_utf8_byte_span': [start, end],
            'json_string_quote_byte_span': item['quote_byte_span'],
            'json_path': json_path(path), 'path_components': list(path),
            'clause_path': json_path(path[:-1]), 'clause': clause,
            'pin': pin, 'record_id': record['id'], 'name': record['name'],
            'definition': record['definition'], 'record': record,
            'dictionary_relations': [
                {'id': i, 'name': resolved['records'][i]['name']}
                for i in record.get('relations', [])
            ],
        })
    need(consumed == set(required_paths), 'a clause concept has no actual tag-2 token')
    need(len(bindings) == 11 and [b['table_index'] for b in bindings] == list(range(11)),
         'concept bindings not complete and consistent')
    return bindings, {
        'actual_tag2_tokens': len(bindings), 'concept_valued_clause_properties': len(required_paths),
        'one_to_one_source_binding': True, 'all_spans_are_UTF8_byte_offsets': True,
        'aliases_derived_from_actual_tag2_tokens': True,
        'binding_summary': [{k: b[k] for k in ['alias', 'token_index', 'table_index',
                                             'expanded_utf8_byte_span', 'json_path', 'record_id', 'name']}
                            for b in bindings],
    }


def application_structure(resolved: dict[str, Any], bindings: list[dict[str, Any]]) -> tuple[dict[str, Any], dict[str, Any]]:
    """Preserve and inventory the supplied grammar; do not execute its message."""
    document = resolved['document']
    expression = document['expression']
    grammar = document['grammar']
    # Structural inventory is independent of English glosses. For this first
    # receiver run it does not presume field/operator names beyond the recipe.
    def inventory(value: Any, path: tuple[Any, ...]) -> list[dict[str, Any]]:
        rows = []
        if isinstance(value, list):
            for i, item in enumerate(value):
                if isinstance(item, dict):
                    rows.append({'json_path': json_path(path + (i,)), 'clause': item})
                    for key, child in item.items():
                        if key == 'then':
                            rows.extend(inventory(child, path + (i, key)))
        elif isinstance(value, dict):
            rows.append({'json_path': json_path(path), 'clause': value})
            if 'then' in value:
                rows.extend(inventory(value['then'], path + ('then',)))
        return rows
    clauses = inventory(expression['clauses'], ('expression', 'clauses'))
    bundle = {
        'source_format': document['format'],
        'context': {k: document['context'][k] for k in ['registry', 'tier', 'head']},
        'grammar': grammar, 'expression': expression,
        'concept_bindings': bindings, 'clause_inventory': clauses,
        'interpretive_boundary': 'Dictionary relations are context, not additional assertions. The message is not executed.',
    }
    return bundle, {'top_level_clauses': len(expression['clauses']),
                    'clause_nodes_including_nested_then': len(clauses),
                    'grammar_preserved': True, 'message_executed': False,
                    'semantic_judgment_is_not_a_cryptographic_check': True}


def write_results(out: Path, png: bytes, wire: bytes, obj: bytes, h: bytes,
                  decoded: dict[str, Any], resolved: dict[str, Any], bindings: list[dict[str, Any]],
                  structured: dict[str, Any]) -> dict[str, Any]:
    files: dict[str, bytes] = {
        'recovered.usl1': obj,
        'recovered.usl.json': pretty_json(decoded['view']),
        'reconstructed.txt': decoded['expanded'],
        'reconstructed-expression.json': decoded['expanded'],
        'checked-snapshot.jsonl': resolved['corpus'],
        'grammar.json': pretty_json(resolved['document']['grammar']),
        'expression.json': pretty_json(resolved['document']['expression']),
        'resolved-expression.json': pretty_json(structured),
        'token-provenance.json': pretty_json(decoded['provenance']),
        'concept-bindings.json': pretty_json(bindings),
        'dictionary-chain-verification.json': pretty_json(resolved['chain']),
        'transport-wire.bin': wire,
        'cvp2-header.bin': h,
    }
    table = {}
    for name, data in files.items():
        (out / name).write_bytes(data)
        table[name] = {'bytes': len(data), 'sha256': sha(data)}
    return table


def run(png_path: Path, out: Path) -> int:
    if out.exists() and any(out.iterdir()):
        raise Reject(f'output directory is not empty; use a new path to retain prior attempts: {out}')
    out.mkdir(parents=True, exist_ok=True)
    evidence = Evidence()
    try:
        need(png_path.stat().st_size <= MAX_PNG_BYTES, 'source file exceeds 80 MiB')
        png = evidence.stage('01 supplied PNG byte count and SHA-256', check_artifact, png_path.read_bytes())
        idat = evidence.stage('02 PNG structure and every chunk CRC', png_container, png)
        rgb = evidence.stage('03 bounded zlib stream and inverse PNG filters', inflate_rgb, idat)
        modules = evidence.stage('04 uniformity of every 4x4 module / every physical pixel', exact_modules, rgb)
        wire = evidence.stage('05 entire canonical non-data frame and RGB24 byte extraction', validate_frame, modules)
        words = evidence.stage('06 deinterleaving and all Reed-Solomon syndromes', rs_verify, wire)
        evidence.stage('07 extra: polynomial-division RS re-encoding', rs_reencode_diagnostic, words)
        obj, h = evidence.stage('08 complete CVP2 header and object integrity', cvp2_object, words)
        decoded = evidence.stage('09 complete USL1 parsing and token provenance', unpack_usl, obj)
        evidence.stage('10 extra: USL1 byte-for-byte re-encoding', reencode_usl, obj, decoded)
        resolved = evidence.stage('11 source-span JSON parsing and embedded dictionary chain', resolve_context, decoded)
        bindings = evidence.stage('12 one-to-one real token / JSON-path concept bindings', bind_concepts, decoded, resolved)
        structured = evidence.stage('13 supplied grammar and expression inventory', application_structure, resolved, bindings)
        evidence.report['files'] = write_results(out, png, wire, obj, h, decoded, resolved, bindings, structured)
        evidence.report['status'] = 'PASS'
        evidence.report['required_check_failures'] = []
        evidence.report['diagnostics']['legacy_checked_fields'] = {
            'status': 'allowed-by-recipe',
            'missing_checked_property_ids': [r['id'] for r in resolved['records'] if 'checked' not in r],
            'explicit_checked_false_ids': [r['id'] for r in resolved['records'] if r.get('checked') is False],
            'corpus_modified': False,
        }
    except Exception as exc:
        evidence.report['status'] = 'FAIL'
        evidence.report['failure'] = {'stage': evidence.current, 'type': type(exc).__name__, 'message': str(exc)}
        (out / 'verification-report.json').write_bytes(pretty_json(evidence.report))
        print(f'FAIL: {evidence.current}: {exc}', file=sys.stderr)
        return 1
    (out / 'verification-report.json').write_bytes(pretty_json(evidence.report))
    lines = []
    for p in sorted(out.iterdir()):
        if p.is_file() and p.name != 'SHA256SUMS.txt':
            lines.append(f'{sha(p.read_bytes())}  {p.name}')
    (out / 'SHA256SUMS.txt').write_text('\n'.join(lines) + '\n', encoding='ascii')
    print('PASS: all supplied intact-frame checks completed.')
    for item in evidence.report['checks']:
        print(f"{item['status']}: {item['stage']}")
    print(f"recovered.usl1: {len(obj)} bytes; SHA-256 {sha(obj)}")
    print(f"reconstructed-expression.json: {len(decoded['expanded'])} bytes; SHA-256 {sha(decoded['expanded'])}")
    print(f'Output: {out}')
    return 0


def main() -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument('png', type=Path, help='original e01-cvp2-rgb24.png')
    parser.add_argument('--out', type=Path, default=Path('cvp2-recovery'))
    args = parser.parse_args()
    try:
        return run(args.png, args.out)
    except (OSError, Reject) as exc:
        print(f'STOPPED: {exc}', file=sys.stderr)
        return 1


if __name__ == '__main__':
    raise SystemExit(main())
