#!/usr/bin/env python3
"""Independent integrity checks for the portable SUBSTELLAR Edition 06 reader.

Requires Python 3.10+ and lxml (``python -m pip install lxml``). It never imports
the reader builder. Default execution is read-only except for its own JSON report.
Use --rebuild only after the maintainer/builder author agrees that source writing
has paused: it invokes the documented builder ONCE and compares its output bytes.
Use --navigation to run the preserved pure-logic Node test in a temporary copy;
this cannot overwrite the preserved historical test receipt. No browser, game,
network, image generation or publication is part of this verifier.

    python sources/verify_edition06.py --navigation --rebuild

The report keeps browser/UI and physical/gameplay claims explicitly NOT RUN.
"""
from __future__ import annotations

import argparse
from collections import Counter, defaultdict
from copy import deepcopy
from datetime import datetime, timezone
from hashlib import sha256
from html.parser import HTMLParser
import json
import os
from pathlib import Path
import re
import shutil
import subprocess
import sys
import tempfile
from urllib.parse import unquote, urlsplit

try:
    from lxml import etree, html
except ImportError:
    raise SystemExit('Missing dependency: lxml. Install with: python -m pip install lxml')


NARRATIVES = [
    'reconciliation_06.txt', 'earth_completion_06.txt',
    'old_valley_completion_06.txt', 'industry_completion_06.txt',
    'advanced_industry_routes_06.txt', 'materials_completion_06.txt',
    'ecology_completion_06.txt', 'tier1_completion_06.txt',
    'moon_completion_06.txt', 'support_vehicles_06.txt', 'orbital_places_06.txt',
]
ALLOWED_BASELINE_CHANGES = {'index.html', 'README.txt', 'sources/edition-manifest.json'}
BUILD_OUTPUTS = [
    'index.html', 'README.txt', 'sources/figures_06.json',
    'sources/visual_catalogue_06.json', 'sources/source_anchor_map_06.json',
    'sources/complete_direction_06.txt', 'sources/edition-manifest.json',
    'sources/reader_build_report_06.json',
]
# Exact corrected originals recorded before the shared scratch reset. These pins
# independently prevent a recovered rejected original from being relabelled as
# the corrected selection merely by changing the selection ledger's hash.
CORRECTED_ORIGINALS = {
    'images/originals/c06-a01.png': '41da6b0facca33511c29dfc5f0b18dc474b88e7f58051c452f845c3aaa6fbe56',
    'images/originals/c06-a02.png': 'cbdd8bf4ce630761c437c9ae7330732f25abde7127453b614e57d48184cdcf48',
}
VOID_TAGS = {'area','base','br','col','embed','hr','img','input','link','meta','param','source','track','wbr'}
SPACE = re.compile(r'\s+')
BOUNDARY = re.compile(r'^(?:#{1,4} |\||[-*] |\d+\. )')
CLASS_SOURCE = 'contains(concat(" ",normalize-space(@class)," ")," source-block ")'
CLASS_CHAPTER = 'contains(concat(" ",normalize-space(@class)," ")," chapter ")'


def norm(value):
    return SPACE.sub(' ', value).strip()


def digest(path):
    return sha256(path.read_bytes()).hexdigest()


def read_json(path):
    return json.loads(path.read_text(encoding='utf-8'))


def parse_html(text):
    return html.fromstring(text, parser=html.HTMLParser(encoding='utf-8', remove_comments=False))


def element_text(element):
    copy = deepcopy(element)
    for decoration in copy.xpath('.//*[contains(concat(" ",normalize-space(@class)," ")," anchor-mark ")]'):
        decoration.drop_tree()
    # Inline links do not introduce spaces before adjacent punctuation. Table
    # cells do introduce semantic boundaries even when tags have no whitespace.
    if copy.tag == 'table':
        return norm(' '.join(''.join(cell.itertext()) for cell in copy.xpath('.//th|.//td')))
    return norm(''.join(copy.itertext()))


def independent_blocks(text):
    """Read the intentionally simple source format independently of the builder."""
    lines = text.splitlines()
    blocks = []
    i = 0
    while i < len(lines):
        line = lines[i].strip()
        if not line:
            i += 1
            continue
        if line.startswith('|'):
            rows = []
            while i < len(lines) and lines[i].strip().startswith('|'):
                cells = [c.strip() for c in lines[i].strip().strip('|').split('|')]
                if not all(re.fullmatch(r':?-+:?', c.replace(' ','')) for c in cells):
                    rows.append(cells)
                i += 1
            cells = [c for row in rows for c in row]
            blocks.append({'kind':'table','text':norm(' '.join(cells)), 'search_text':' | '.join(cells)})
            continue
        heading = re.match(r'^(#{1,4}) (.+)$', line)
        if heading:
            blocks.append({'kind':'heading','level':len(heading[1]),'text':heading[2]})
            i += 1
            continue
        bullet = re.match(r'^[-*] (.+)$',line)
        number = re.match(r'^\d+\. (.+)$',line)
        if bullet or number:
            blocks.append({'kind':'bullet' if bullet else 'number','text':(bullet or number)[1]})
            i += 1
            continue
        paragraph = [line]
        i += 1
        while i < len(lines) and lines[i].strip() and not BOUNDARY.match(lines[i].strip()):
            paragraph.append(lines[i].strip())
            i += 1
        blocks.append({'kind':'paragraph','text':' '.join(paragraph)})
    return blocks


class RawSourceBlocks(HTMLParser):
    """Capture the literal old/new element strings, beyond normalized DOM text."""
    def __init__(self, text):
        super().__init__(convert_charrefs=False)
        self.text = text
        self.starts = [0] + [m.end() for m in re.finditer('\n',text)]
        self.stack = []
        self.blocks = defaultdict(list)
        self.feed(text)
        self.close()

    def position(self):
        line, column = self.getpos()
        return self.starts[line-1] + column

    def handle_starttag(self, tag, attributes):
        if tag in VOID_TAGS:
            return
        attributes = dict(attributes)
        key = None
        if 'source-block' in (attributes.get('class') or '').split():
            key = (attributes.get('data-source'),attributes.get('data-block'))
        self.stack.append((tag,self.position(),key))

    def handle_startendtag(self, tag, attributes):
        pass

    def handle_endtag(self, tag):
        match = next((i for i in range(len(self.stack)-1,-1,-1) if self.stack[i][0]==tag),None)
        if match is None:
            return
        _,start,key = self.stack[match]
        if key is not None:
            end = self.text.find('>', self.position())+1
            self.blocks[key].append(self.text[start:end])
        del self.stack[match:]


class Verifier:
    def __init__(self, root):
        self.root = root
        self.checks = []
        self.counts = {}
        self.artifacts = {}
        self.notes = []

    def check(self, name, passed, details=None):
        entry = {'name':name,'status':'PASS' if passed else 'FAIL'}
        if details is not None:
            entry['details'] = details
        self.checks.append(entry)
        return bool(passed)

    def skipped(self, name, reason):
        self.checks.append({'name':name,'status':'NOT RUN','reason':reason})

    def required_json(self, relative):
        return read_json(self.root/relative)

    def verify_preservation(self, old_doc, new_doc, old_text, new_text):
        inventory = self.required_json('sources/baseline_05_inventory.json')
        baseline_manifest = self.required_json('sources/edition05_manifest_preserved.json')
        by_path = {r['path']:r for r in inventory}
        missing = []
        changed = []
        allowed_changed = []
        for item in inventory:
            file = self.root/item['path']
            if not file.is_file():
                missing.append(item['path'])
                continue
            same = file.stat().st_size==item['bytes'] and digest(file)==item['sha256']
            if not same:
                (allowed_changed if item['path'] in ALLOWED_BASELINE_CHANGES else changed).append(item['path'])
        self.check('All 191 baseline files retained; only three current edition outputs may differ',
                   len(inventory)==191 and not missing and not changed,
                   {'missing':missing,'unauthorized_changes':changed,'permitted_changed_outputs':allowed_changed})
        sources = baseline_manifest['source_content']
        source_paths = ['sources/'+x['file'].removeprefix('sources/') for x in sources]
        source_mismatches = [p for p in source_paths if p not in by_path or not (self.root/p).is_file() or digest(self.root/p)!=by_path[p]['sha256']]
        art_paths = [x['path'] for x in inventory if x['path'].startswith('images/')]
        art_mismatches = [p for p in art_paths if not (self.root/p).is_file() or digest(self.root/p)!=by_path[p]['sha256']]
        self.check('All 11 baseline narrative files byte-identical',len(source_paths)==11 and not source_mismatches,source_mismatches)
        self.check('All 93 baseline artwork files byte-identical',len(art_paths)==93 and not art_mismatches,art_mismatches)
        self.check('Preserved reader matches baseline index hash',digest(self.root/'sources/reader_05_preserved.html')==by_path['index.html']['sha256'])
        self.check('Preserved manifest matches original edition manifest hash',digest(self.root/'sources/edition05_manifest_preserved.json')==by_path['sources/edition-manifest.json']['sha256'])
        self.counts.update(baseline_files=len(inventory),baseline_narrative_files=len(source_paths),baseline_artwork_files=len(art_paths))

        old_ids = {x.get('id') for x in old_doc.xpath('//*[@id]')}
        new_ids = [x.get('id') for x in new_doc.xpath('//*[@id]')]
        missing_ids = sorted(old_ids-set(new_ids))
        duplicates = {k:v for k,v in Counter(new_ids).items() if v>1}
        self.check('Every old DOM ID survives',not missing_ids,{'old_id_count':len(old_ids),'missing':missing_ids})
        self.check('Reader DOM IDs are unique',not duplicates,duplicates)
        self.counts.update(old_dom_ids=len(old_ids),current_dom_ids=len(new_ids))

        old_blocks = old_doc.xpath(f'//*[{CLASS_SOURCE}]')
        new_blocks = new_doc.xpath(f'//*[{CLASS_SOURCE}]')
        def grouped(nodes):
            result = defaultdict(list)
            for node in nodes:
                result[(node.get('data-source'),node.get('data-block'))].append(node)
            return result
        old_group,new_group = grouped(old_blocks),grouped(new_blocks)
        duplicate_block_keys = [list(k) for k,v in new_group.items() if len(v)!=1]
        text_changed=[]
        markup_changed=[]
        for key,nodes in old_group.items():
            current = new_group.get(key,[])
            if len(nodes)!=1 or len(current)!=1 or element_text(nodes[0])!=element_text(current[0]):
                text_changed.append(list(key))
            elif etree.tostring(nodes[0],encoding='utf-8',with_tail=False)!=etree.tostring(current[0],encoding='utf-8',with_tail=False):
                markup_changed.append(list(key))
        self.check('All source/block keys remain unique',not duplicate_block_keys,duplicate_block_keys)
        self.check('All 2,000 old source blocks retain their exact rendered text',len(old_blocks)==2000 and not text_changed,text_changed)
        self.check('Old source-block DOM markup remains unchanged',not markup_changed,markup_changed)
        raw_old,raw_new = RawSourceBlocks(old_text).blocks,RawSourceBlocks(new_text).blocks
        raw_changed = [list(k) for k,v in raw_old.items() if v!=raw_new.get(k)]
        self.check('All 2,000 old source-block literal element strings remain unchanged',sum(map(len,raw_old.values()))==2000 and not raw_changed,raw_changed)
        self.counts.update(baseline_source_blocks=len(old_blocks),current_source_blocks=len(new_blocks))

        def executable(doc):
            return [x for x in doc.xpath('//script') if x.get('type','').lower() not in ('application/json','application/ld+json')]
        old_scripts,new_scripts = executable(old_doc),executable(new_doc)
        def script_identity(node):
            return (tuple(sorted(node.attrib.items())),node.text or '')
        old_script_counts,new_script_counts = Counter(map(script_identity,old_scripts)),Counter(map(script_identity,new_scripts))
        self.check('Every original executable script remains unchanged',not (old_script_counts-new_script_counts),{'old_script_count':len(old_scripts),'current_script_count':len(new_scripts)})
        script_text='\n'.join(x.text or '' for x in old_scripts)
        keys=sorted(set(re.findall(r'''["'](substellar[^"']*)["']''',script_text)))
        current_script='\n'.join(x.text or '' for x in new_scripts)
        self.check('Original storage-key literals survive',all(key in current_script for key in keys),keys)
        return old_group,new_group

    def verify_json_and_xml(self, doc):
        failures=[]
        json_count=0
        xml_count=0
        for file in sorted(self.root.rglob('*.json')):
            try: read_json(file)
            except Exception as error: failures.append({'file':str(file.relative_to(self.root)),'error':str(error)})
            json_count+=1
        for file in sorted(self.root.rglob('*.svg')):
            try: etree.parse(str(file),etree.XMLParser(resolve_entities=False,no_network=True))
            except Exception as error: failures.append({'file':str(file.relative_to(self.root)),'error':str(error)})
            xml_count+=1
        self.check('All metadata JSON and SVG XML parse',not failures,{'json_files':json_count,'svg_files':xml_count,'failures':failures})
        embedded={}
        errors=[]
        for script in doc.xpath('//script[@type="application/json" or @type="application/ld+json"]'):
            try: embedded[script.get('id')]=json.loads(script.text or '')
            except Exception as error: errors.append({'id':script.get('id'),'error':str(error)})
        self.check('All embedded JSON parses',not errors,{'blocks':len(embedded),'failures':errors})
        self.counts.update(metadata_json_files=json_count,svg_files=xml_count,embedded_json_blocks=len(embedded))
        return embedded

    def verify_narratives(self, doc, old_group, new_group, embedded):
        search=embedded.get('search-data',[])
        old_search=json.loads(parse_html((self.root/'sources/reader_05_preserved.html').read_text()).get_element_by_id('search-data').text)
        old_counter=Counter(json.dumps(x,sort_keys=True,ensure_ascii=False) for x in old_search)
        new_counter=Counter(json.dumps(x,sort_keys=True,ensure_ascii=False) for x in search)
        lost=sum((old_counter-new_counter).values())
        self.check('Original search records remain present unchanged',lost==0,{'old_records':len(old_search),'lost':lost})
        groups=defaultdict(list)
        for key,nodes in new_group.items():
            if key not in old_group:
                groups[key[0]].extend(nodes)
        ids={x.get('id') for x in doc.xpath('//*[@id]')}
        sidebar=doc.xpath('//*[@id="sidebar"]')
        sidebar_links=defaultdict(list)
        for link in sidebar[0].xpath('.//a[starts-with(@href,"#")]') if sidebar else []:
            sidebar_links[unquote(link.get('href')[1:])].append(link)
        claimed_keys=set()
        expected_section_ids=set()
        literal_c06_definitions=set()
        new_chapter_ids=set()
        source_reports=[]
        search_failures=[]
        block_failures=[]
        nav_failures=[]
        total_expected=0
        chapter_count=0
        for filename in NARRATIVES:
            source=self.root/'sources'/filename
            expected=independent_blocks(source.read_text())
            total_expected+=len(expected)
            chapter_count+=sum(x['kind']=='heading' and x.get('level')==1 for x in expected)
            candidates=[key for key,nodes in groups.items() if key not in claimed_keys and nodes and element_text(nodes[0])==norm(expected[0]['text'])]
            if len(candidates)!=1:
                block_failures.append({'file':filename,'error':'Expected exactly one matching new data-source group','candidates':candidates})
                continue
            key=candidates[0]
            claimed_keys.add(key)
            rendered=groups[key]
            if len(rendered)!=len(expected):
                block_failures.append({'file':filename,'expected_blocks':len(expected),'rendered_blocks':len(rendered)})
            source_reports.append({'file':filename,'key':key,'blocks':len(expected),'chapters':sum(x['kind']=='heading' and x.get('level')==1 for x in expected),'sha256':digest(source)})
            for position,(block,node) in enumerate(zip(expected,rendered)):
                if norm(block['text'])!=element_text(node):
                    block_failures.append({'file':filename,'block':position,'id':node.get('id'),'error':'Source and rendered text differ','source_excerpt':block['text'][:160],'rendered_excerpt':element_text(node)[:160]})
                targets={node.get('id')}
                chapter=next((a for a in node.iterancestors() if 'chapter' in (a.get('class') or '').split()),None)
                if block['kind']=='heading' and block.get('level')==1:
                    if chapter is not None:
                        targets.add(chapter.get('id'))
                        new_chapter_ids.add(chapter.get('id'))
                        if chapter.get('id') not in sidebar_links:
                            nav_failures.append({'chapter':chapter.get('id'),'error':'No sidebar chapter link'})
                    else:
                        nav_failures.append({'block':node.get('id'),'error':'Source chapter has no chapter container'})
                elif block['kind']=='heading':
                    expected_section_ids.add(node.get('id'))
                    if node.get('id') not in sidebar_links:
                        nav_failures.append({'section':node.get('id'),'error':'No sidebar section link'})
                    ident=re.match(r'(C06-[A-Z0-9-]+)\b',block['text'])
                    if ident:
                        literal_c06_definitions.add(ident[1])
                        if ident[1] not in ids:
                            nav_failures.append({'section':ident[1],'error':'Missing stable section anchor'})
                        if ident[1] not in sidebar_links:
                            nav_failures.append({'section':ident[1],'error':'No sidebar section link'})
                matching=[x for x in search if x.get('id') in targets and x.get('source')==key and norm(x.get('text',''))==norm(block.get('search_text',block['text']))]
                if len(matching)!=1:
                    search_failures.append({'file':filename,'block':position,'target_candidates':sorted(t for t in targets if t),'matching_entries':len(matching)})
        unclaimed=sorted(set(groups)-claimed_keys)
        self.check('All 11 new narratives are rendered completely, in source-block order, exactly once',len(source_reports)==11 and not block_failures and not unclaimed,{'sources':source_reports,'failures':block_failures,'unclaimed_source_groups':unclaimed})
        self.check('Every new narrative block has exactly one complete search record',not search_failures,{'expected_blocks':total_expected,'failures':search_failures})
        self.check('All 18 new chapters and their C06 sections have sidebar links',chapter_count==18 and len(new_chapter_ids)==18 and not nav_failures,{'chapters':len(new_chapter_ids),'sections':len(expected_section_ids),'failures':nav_failures})
        self.counts.update(new_narrative_files=len(source_reports),new_source_blocks=total_expected,new_chapters=len(new_chapter_ids),new_section_anchors=len(expected_section_ids),literal_c06_source_heading_ids=len(literal_c06_definitions),search_records=len(search))
        self.artifacts['new_sources']=source_reports
        return expected_section_ids

    def verify_figures_and_anchors(self, doc, embedded):
        ids={x.get('id') for x in doc.xpath('//*[@id]')}
        catalog=self.required_json('sources/visual_catalogue_06.json')
        records=catalog['records'] if isinstance(catalog,dict) else catalog
        embedded_visual=embedded.get('visual-data',{})
        embedded_records=embedded_visual.get('records',[]) if isinstance(embedded_visual,dict) else embedded_visual
        self.check('84 distinct catalogue visuals match the embedded catalogue',len(records)==84 and len({x['id'] for x in records})==84 and records==embedded_records,{'records':len(records),'embedded_records':len(embedded_records)})
        hrefs={unquote(x.get('href')[1:]) for x in doc.xpath('//a[starts-with(@href,"#")]')}
        record_map={x['id']:x for x in records}
        failures=[]
        for record in records:
            target=record.get('target') or record.get('anchor') or record.get('href','').removeprefix('#')
            if target not in ids or target not in hrefs:
                failures.append({'id':record['id'],'target':target,'error':'Missing figure anchor or discovery link'})
                continue
            node=doc.get_element_by_id(target)
            images=node.xpath('.//img[@src]')
            if node.tag=='img': images=[node]+images
            inline_svg=node.xpath('.//svg')
            svg_links=[a.get('href') for a in node.xpath('.//a[@href]') if urlsplit(a.get('href')).path.lower().endswith('.svg')]
            if not images and not (inline_svg and any((self.root/unquote(urlsplit(link).path)).is_file() for link in svg_links)):
                failures.append({'id':record['id'],'target':target,'error':'Visual has neither a display image nor inline SVG with an existing full-size source'})
            for image in images:
                if not (self.root/unquote(urlsplit(image.get('src')).path)).is_file():
                    failures.append({'id':record['id'],'error':'Visual file missing','file':image.get('src')})
        self.check('Every one of 84 visuals is reachable with an existing image file',not failures,failures)

        figure_data=self.required_json('sources/figures_06.json')
        figures=figure_data.get('figures',figure_data.get('records',[])) if isinstance(figure_data,dict) else figure_data
        expected={f'C06-D{i:02d}' for i in range(1,14)}|{f'C06-A{i:02d}' for i in range(1,13)}
        actual={f['id'] for f in figures}
        failures=[]
        for figure in figures:
            ident=figure['id']
            if ident not in record_map:
                failures.append({'id':ident,'error':'Missing visual catalogue entry'})
            for key in ['file','original','web','display']:
                value=figure.get(key)
                if value and not (self.root/value).is_file():
                    failures.append({'id':ident,'error':'Missing figure file','file':value})
            for key,hash_key in [('file','sha256_file'),('original','sha256_original')]:
                value,expected_hash=figure.get(key),figure.get(hash_key)
                if value and expected_hash and (self.root/value).is_file() and digest(self.root/value)!=expected_hash:
                    failures.append({'id':ident,'error':'Figure register file hash mismatch','file':value})
            for section in figure.get('section_ids',[]):
                if section not in ids: failures.append({'id':ident,'error':'Missing linked section anchor','section':section})
        self.check('All 25 new figure identities/files/context anchors resolve',actual==expected and len(figures)==25 and not failures,{'missing_ids':sorted(expected-actual),'unexpected_ids':sorted(actual-expected),'failures':failures})

        section_refs=set()
        def visit(value,key=None):
            if isinstance(value,dict):
                for k,v in value.items(): visit(v,k)
            elif isinstance(value,list):
                for item in value: visit(item,key)
            elif isinstance(value,str) and key in {'section_id','section_ids','text_section_ids','primary_section_id'} and value.startswith('C06-'):
                section_refs.add(value)
        for file in (self.root/'sources').glob('*_06.json'):
            if file.name not in ('verification_06.json',): visit(read_json(file))
        missing=sorted(section_refs-ids)
        self.check('All Edition06 metadata C06 section references resolve',not missing,{'unique_section_references':len(section_refs),'missing':missing})
        anchor_map=self.required_json('sources/source_anchor_map_06.json')
        self.artifacts['source_anchor_map_schema']=list(anchor_map) if isinstance(anchor_map,dict) else 'array'
        mapping_failures=[]
        for group in ['canonical_symbol_anchors','historical_symbol_anchors_05','edition06_source_block_targets','figure_anchors']:
            for symbol,target in anchor_map.get(group,{}).items():
                if target.removeprefix('#') not in ids:
                    mapping_failures.append({'group':group,'symbol':symbol,'target':target})
        for chapter in anchor_map.get('chapters',[]):
            for target in [chapter]+chapter.get('sections',[]):
                if target['id'] not in ids:
                    mapping_failures.append({'group':'chapters','target':target['id']})
        self.check('Every source-map symbol/block/chapter/section/figure target resolves',not mapping_failures,mapping_failures)
        self.check('Source-anchor map is bound to the current reader hash',anchor_map.get('reader_sha256')==digest(self.root/'index.html'))
        self.counts.update(visual_entries=len(records),new_figures=len(figures))

    def verify_complete_source(self):
        old_manifest=self.required_json('sources/edition05_manifest_preserved.json')
        paths=['sources/'+s['file'].removeprefix('sources/') for s in old_manifest['source_content']]
        paths+=['sources/'+s for s in NARRATIVES]
        combined=(self.root/'sources/complete_direction_06.txt').read_text()
        failures=[]
        for relative in paths:
            text=(self.root/relative).read_text().strip()
            count=combined.count(text)
            if count!=1: failures.append({'file':relative,'complete_source_occurrences':count})
        self.check('Complete plain-text edition contains all 22 full narrative sources exactly once',len(paths)==22 and not failures,failures)
        current_manifest=self.required_json('sources/edition-manifest.json')
        manifest_sources=current_manifest.get('source_content',[])
        failures=[]
        for source in manifest_sources:
            relative='sources/'+source['file'].removeprefix('sources/')
            if not (self.root/relative).is_file() or digest(self.root/relative)!=source.get('sha256'):
                failures.append({'file':relative,'error':'Current manifest source hash mismatch'})
        self.check('Current manifest names and hashes all 22 narrative sources',len(manifest_sources)==22 and {s['file'].removeprefix('sources/') for s in manifest_sources}=={p.removeprefix('sources/') for p in paths} and not failures,failures)

    def verify_local_links(self, doc):
        ids={x.get('id') for x in doc.xpath('//*[@id]')}
        missing=[]
        checked=set()
        external=set()
        html_cache={self.root/'index.html':ids}
        for element in doc.xpath('//*[@href or @src]'):
            for attribute in ('href','src'):
                raw=element.get(attribute)
                if raw is None: continue
                value=urlsplit(raw)
                if value.scheme or value.netloc:
                    external.add(raw)
                    continue
                path=unquote(value.path)
                fragment=unquote(value.fragment)
                if not path:
                    if fragment and fragment not in ids:
                        missing.append({'attribute':attribute,'value':raw,'error':'Missing reader fragment'})
                    checked.add(raw)
                    continue
                target=(self.root/path).resolve()
                if not target.is_relative_to(self.root) or path.startswith('/'):
                    missing.append({'attribute':attribute,'value':raw,'error':'Nonportable local path'})
                    continue
                if not target.exists():
                    missing.append({'attribute':attribute,'value':raw,'error':'Missing file'})
                elif fragment and target.suffix.lower() in {'.html','.htm'}:
                    if target not in html_cache:
                        html_cache[target]={x.get('id') for x in parse_html(target.read_text()).xpath('//*[@id]')}
                    if fragment not in html_cache[target]:
                        missing.append({'attribute':attribute,'value':raw,'error':'Missing linked HTML fragment'})
                checked.add(raw)
        self.check('Every local href/src target exists and HTML fragments resolve',not missing,{'distinct_local_targets':len(checked),'failures':missing})
        self.counts['external_links_not_fetched']=len(external)

    def verify_image_budget(self):
        ledger=self.required_json('sources/image_generation_06.json')
        calls=ledger['calls']
        budget=ledger['budget']
        batches=Counter(x['batch'] for x in calls)
        selected=[x for x in calls if x.get('selected')]
        rejected=[x for x in calls if not x.get('selected') and 'rejected' in x.get('status','').lower()]
        failed=[x for x in calls if not x.get('selected') and 'failed' in x.get('status','').lower()]
        self.check('Generation ledger is complete and within the agreed 16-call / four-per-batch cap',
                   len(calls)==15 and budget['calls_used_in_this_programme']==15 and budget['maximum_total_calls']==16
                   and budget['maximum_calls_per_batch']==4 and max(batches.values(),default=0)<=4
                   and len(selected)==12 and len(rejected)==2 and len(failed)==1
                   and [x['call'] for x in calls]==list(range(1,16)),
                   {'calls':len(calls),'selected':len(selected),'rejected':len(rejected),'failed':len(failed),'batches':dict(batches)})
        failures=[]
        for call in selected:
            original=self.root/call['original']
            if not original.is_file() or digest(original)!=call['sha256']:
                failures.append({'call':call['call'],'id':call['asset_id'],'error':'Selected original/hash mismatch'})
            if not (self.root/call['display']).is_file():
                failures.append({'call':call['call'],'id':call['asset_id'],'error':'Missing selected display export'})
        self.check('All 12 selected originals match their ledger hashes and display exports exist',not failures,failures)
        rejected_hashes={c.get('rejected_output_sha256') for c in rejected}
        valid_hashes=all(isinstance(h,str) and re.fullmatch('[0-9a-f]{64}',h) for h in rejected_hashes)
        matches=[]
        for file in self.root.rglob('*'):
            if file.is_file() and digest(file) in rejected_hashes:
                matches.append(str(file.relative_to(self.root)))
        self.check('Known rejected realistic-image hashes are absent from every package file',valid_hashes and not matches,{'known_rejected_hashes':sorted(h for h in rejected_hashes if h),'matches':matches})
        corrected=[]
        for relative,expected in CORRECTED_ORIGINALS.items():
            actual=digest(self.root/relative) if (self.root/relative).is_file() else None
            if actual!=expected: corrected.append({'file':relative,'expected':expected,'actual':actual})
        self.check('A01/A02 match the recorded corrected Minecraft-style originals',not corrected,corrected)
        self.counts.update(generation_calls=len(calls),selected_new_concepts=len(selected),rejected_calls=len(rejected),failed_calls=len(failed))

    def navigation(self, requested):
        if not requested:
            self.skipped('Preserved navigation pure-logic test','Run with --navigation to execute in an isolated temporary copy.')
            return
        executable=shutil.which('node') or os.environ.get('CODEX_PRIMARY_RUNTIME_NODE')
        if not executable or not Path(executable).is_file():
            self.skipped('Preserved navigation pure-logic test','Node executable is unavailable in this environment.')
            return
        with tempfile.TemporaryDirectory(prefix='substellar-nav-verify-') as folder:
            directory=Path(folder)
            for name in ['test_reader_navigation.js','substellar_reader_navigation.js']:
                shutil.copyfile(self.root/'sources'/name,directory/name)
            run=subprocess.run([executable,'test_reader_navigation.js'],cwd=directory,capture_output=True,text=True,timeout=60)
            report=read_json(directory/'reader_navigation_test_report.json') if (directory/'reader_navigation_test_report.json').exists() else None
            result=run.returncode==0 and report is not None and report.get('passed') is True
            details={'returncode':run.returncode,'test_count':len(report.get('tests',[])) if report else 0,'report':report,'stderr':run.stderr[:3000]}
            self.check('Preserved navigation pure-logic test in isolated copy',result,details)

    def rebuild(self, requested):
        if not requested:
            self.skipped('One deterministic builder rerun','Not requested; use --rebuild only after authoring is paused.')
            return
        missing=[f for f in BUILD_OUTPUTS if not (self.root/f).is_file()]
        if missing:
            self.check('One deterministic builder rerun',False,{'error':'Expected outputs must exist before the single rerun','missing':missing})
            return
        before={f:digest(self.root/f) for f in BUILD_OUTPUTS}
        run=subprocess.run([sys.executable,str(self.root/'sources/build_edition06.py')],cwd=self.root,capture_output=True,text=True,timeout=120)
        after={f:digest(self.root/f) if (self.root/f).is_file() else None for f in BUILD_OUTPUTS}
        changed={f:{'before':before[f],'after':after[f]} for f in before if before[f]!=after[f]}
        self.check('One deterministic builder rerun',run.returncode==0 and not changed,
                   {'executions':1,'returncode':run.returncode,'compared_outputs':BUILD_OUTPUTS,'changed':changed,'stdout':run.stdout[-4000:],'stderr':run.stderr[-4000:]})

    def run(self, args):
        # Rebuild first, then independently validate the resulting package.
        self.rebuild(args.rebuild)
        old_text=(self.root/'sources/reader_05_preserved.html').read_text(encoding='utf-8')
        new_text=(self.root/'index.html').read_text(encoding='utf-8')
        old_doc,new_doc=parse_html(old_text),parse_html(new_text)
        old_group,new_group=self.verify_preservation(old_doc,new_doc,old_text,new_text)
        embedded=self.verify_json_and_xml(new_doc)
        self.verify_narratives(new_doc,old_group,new_group,embedded)
        self.verify_figures_and_anchors(new_doc,embedded)
        self.verify_complete_source()
        self.verify_local_links(new_doc)
        self.verify_image_budget()
        self.navigation(args.navigation)


def main():
    parser=argparse.ArgumentParser(description=__doc__,formatter_class=argparse.RawDescriptionHelpFormatter)
    parser.add_argument('--root',type=Path,default=Path(__file__).resolve().parent.parent)
    parser.add_argument('--output',type=Path,help='Default: ROOT/sources/verification_06.json')
    parser.add_argument('--navigation',action='store_true',help='Run existing pure Node tests in an isolated temporary copy')
    parser.add_argument('--rebuild',action='store_true',help='Run builder exactly once; coordinate a pause in authoring first')
    args=parser.parse_args()
    root=args.root.resolve()
    verifier=Verifier(root)
    try:
        verifier.run(args)
    except Exception as error:
        verifier.check('Verifier completed all planned static checks',False,{'exception_type':type(error).__name__,'error':str(error)})
    failures=[x for x in verifier.checks if x['status']=='FAIL']
    report={
        'edition':'selected-completion-06',
        'verifier':'Independent reader/package integrity verification',
        'verified_at_utc':datetime.now(timezone.utc).isoformat(),
        'status':'PASS' if not failures else 'FAIL',
        'reader_sha256':digest(root/'index.html') if (root/'index.html').exists() else None,
        'verifier_sha256':digest(Path(__file__)),
        'scope':'Baseline bytes and exact source blocks; new narrative/search/navigation/figure coverage; local links and metadata; generation accounting; optional single deterministic rebuild and pure Node logic.',
        'counts':verifier.counts,
        'checks':verifier.checks,
        'artifacts':verifier.artifacts,
        'failure_count':len(failures),
        'not_run':['Browser or rendered UI verification','Actual sticky/follow behavior in a browser','Keyboard/focus, screen-reader, visual/mobile or print execution','Website production/deployment gates','Game/mod/runtime/model-fit checks','External URL availability'],
        'limits':['Static sidebar links and preserved navigation logic do not prove responsive geometry or live browser behavior.','This verifier does not generate artwork, modify baseline sources, publish content or approve mod development.','Hash exclusion proves the known rejected original bytes are absent; visual art review remains a separately recorded maintainer task.'],
    }
    output=args.output or root/'sources/verification_06.json'
    output.parent.mkdir(parents=True,exist_ok=True)
    output.write_text(json.dumps(report,indent=2,ensure_ascii=False)+'\n',encoding='utf-8')
    print(json.dumps({'status':report['status'],'checks':len(verifier.checks),'failures':len(failures),'report':str(output),'counts':verifier.counts},ensure_ascii=False))
    if failures:
        for failure in failures:
            print(json.dumps(failure,ensure_ascii=False))
    return 1 if failures else 0


if __name__=='__main__':
    raise SystemExit(main())
