#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""Mechanical validator for the Agentic SDLC family — the shared core.

This module is the SPINE: it is authored once here and shipped verbatim in every
distribution of the family (code, knowledge, marketing). It carries nothing that
belongs to a single domain. Each distribution ships a thin entry point beside it —
`sdlc_check.py` for code, `mkt_check.py` for marketing — which names the domain it
implements and delegates everything else here. A drift guard fails CI when the
copies diverge, so do not fork this file: fix it once and copy it.

It is also runnable on its own (`python sdlc_core.py check`) for the same reason
the entry points are thin: the behaviour is here, not in them.

Commands:
  check      single closure gate: validate + stale in one command (exit 1 if either fails)
  validate   verify the structural coherence of the docs root (default ai_docs/; exit 1 on
             errors; --strict also fails on warnings or on a missing docs root, for CI)
  index      regenerate the generated indexes: strategic/features_history.md (from the
             frontmatter of ANALYSIS_*.md files) and ai_docs/INDEX.md (manifest of canonical docs)
  stale      list areas modified after the last analysis recorded in audit_plan.md (exit 1 if any)
  mark       record paths as ANALYZED with the current reference (git hash, else UTC timestamp)
  gate       PreToolUse hook: block writes on protected paths without an IN_PROGRESS ANALYSIS (exit 2)

Hybrid/devPNT mode: pass --hybrid explicitly on check/stale (skips audit-plan
staleness, delegated to devPNT/KL) and on gate (also unlocks when an approved
E-TDD shadow, solutions/SHADOW_*tdd*.md, exists).

Canonical language is English. Legacy Italian frontmatter keys (stato, livello,
data_inizio, data_fine) and section headings are still accepted for existing projects,
but are deprecated: new documents should use the English forms.

Standard library only (Python >= 3.8). Windows and POSIX compatible.
"""
import argparse
import hashlib
import json
import os
import re
import subprocess
import sys
from datetime import datetime, timedelta, timezone
from pathlib import Path

VALID_STATES = {"PLANNED", "IN_PROGRESS", "COMPLETED", "CANCELLED"}
VALID_LEVELS = {"L1", "L2", "L3", "SPIKE"}
VISION_FILES = ("project_vision.md", "roadmap.md", "principles.md")
SKIP_DIRS = {".git", ".hg", ".svn", "node_modules", "__pycache__", ".venv", "venv",
             "dist", "build", ".idea", ".vs"}  # the docs root is excluded separately: its
# name is resolved per invocation, so it cannot live in a frozen set (see docs_dir()).
INDEX_HEADER = ("<!-- GENERATED by sdlc_check.py index - do not edit by hand. "
                "Source of truth: frontmatter of the ANALYSIS_*.md files -->")
def manifest_header():
    return ("<!-- GENERATED by sdlc_check.py index - do not edit by hand. "
            f"Source of truth: the headers of the canonical documents in {docs_dir()}/. -->")
# Directories whose .md files are durable canonical documents: manifested in INDEX.md.
# audit/ and solutions/ stay discovery-by-grep (session / process artifacts), not manifested.
MANIFEST_DIRS = ("vision", "reference", "architecture", "functional", "strategic")
# Recognized states: canonical docs (CURRENT/SUPERSEDED/...), vision (DRAFT/APPROVED),
# ADR (Accepted/Proposed/Rejected). Union, to avoid false warnings on conventions in use.
CANONICAL_STATES = {"CURRENT", "SUPERSEDED", "DRAFT", "DEPRECATED",
                    "APPROVED", "ACCEPTED", "PROPOSED", "REJECTED"}
GENERATED_DOCS = {"features_history.md", "INDEX.md"}  # generated: never manifest entries
MTIME_GRACE = timedelta(seconds=2)
def guide_index_header():
    return ("<!-- GENERATED by sdlc_check.py index - do not edit by hand. "
            f"Source of truth: the headers of the GUIDE_*.md files in {docs_dir()}/reference/. -->")
GUIDE_PROVENANCE_KEYS = ("source", "distilled_from", "source_hash")  # source_version optional
# a guide section is "covered" when it carries a source marker or an explicit gap marker
GUIDE_MARKER_RE = re.compile(r"\[(?:source:[^\]]+|not covered by source)\]")
# Agent-global KB (Feature B unit 2): ONE client-agnostic root under home.
# AGENTIC_SDLC_KB_ROOT env var is a TEST/CI seam only (scenario battery must
# not touch the real user KB); the documented product path is fixed.
DEFAULT_KB_ROOT = Path(os.environ.get("AGENTIC_SDLC_KB_ROOT", "")) if os.environ.get("AGENTIC_SDLC_KB_ROOT") else Path.home() / ".agentic-sdlc"
# Subagent Execution (Feature A): a PLAN_[feature].md task must carry these keys,
# plus at least one of paths/produces (checked separately in cmd_plan).
PLAN_TASK_REQUIRED = ("id", "title", "verify")

# Deprecated Italian frontmatter keys, mapped to the canonical English ones.
LEGACY_KEYS = {"stato": "status", "livello": "level",
               "data_inizio": "start_date", "data_fine": "end_date"}

# Architect pass (F-020): the Capability Ledger is due for ACTIVE L3 analyses
# born on/after the day the pass shipped. Grandfathering by start_date -- an
# in-flight analysis from before the pass existed never nags (same lazy-convert
# doctrine as the pre-1.17 narrative handoff).
ARCHITECT_PASS_EPOCH = "2026-07-28"
# Design-review gate (F-021): an L3 started on/after this date owes a REVIEW_LOG
# row. Same grandfathering discipline as the pass above -- never nag work that
# predates the rule.
DESIGN_REVIEW_EPOCH = "2026-07-28"
def review_log_rel():
    return f"{docs_dir()}/audit/reviews/REVIEW_LOG.md"
# Component Map 'Where' refs: a dotted token counts as a path only with one of
# these suffixes. Deliberately a closed list -- a generic ".\w{1,5}$" turns
# `app.core`, `OrderStore.save` and `1.18.0` into "the map is rotting".
FILE_SUFFIXES = ("md", "py", "js", "mjs", "cjs", "ts", "tsx", "jsx", "json", "yaml",
                 "yml", "toml", "ini", "cfg", "sh", "bat", "ps1", "go", "rs", "java",
                 "kt", "rb", "php", "cs", "swift", "c", "h", "cpp", "hpp", "sql",
                 "css", "scss", "html", "vue", "svelte", "tf", "proto", "txt")

# ANALYSIS sections: (canonical English heading, legacy Italian heading).
SECURITY_SECTION = ("## Security", "## Sicurezza")
ANALYSIS_SECTIONS = (
    ("## Objective", "## Obiettivo"),
    ("## Feature Vision", "## Vision della Feature"),
    ("## Impact", "## Impatto"),
    ("## Action Plan", "## Piano d'Azione"),
    ("## Test Strategy", "## Strategia di Test"),
    ("## Diary", "## Diario"),
)

# ------------------------------------------------------------------- docs root
# `ai_docs/` is the ONE surviving documentation root and the default everywhere.
# The name is a parameter for exactly two reasons, both temporary by nature: a
# project that predates the convention has to be READ before it can be migrated,
# and the migration tool has to see both sides. It is not an invitation to keep a
# second permanent root -- which is why `init.js` has no rename knob.
DEFAULT_DOCS_DIR = "ai_docs"
DOCS_DIR_CANDIDATES = ("ai_docs", "mkt_docs")
DOCS_DIR_ENV = "AGENTIC_SDLC_DOCS_DIR"  # test/CI seam, read per invocation
_DOCS_DIR = {"name": DEFAULT_DOCS_DIR}


def docs_dir():
    """The resolved documentation root NAME for this invocation."""
    return _DOCS_DIR["name"]


def set_docs_dir(name):
    _DOCS_DIR["name"] = name or DEFAULT_DOCS_DIR


def ai_path(root, *parts):
    """A path inside the resolved documentation root."""
    return Path(root).joinpath(docs_dir(), *parts)


def resolve_docs_dir(args=None, start=None):
    """Explicit beats guessed, exactly like --hybrid.

    Order: --docs-dir, then the env seam (read now, not at import, or it could not
    be varied by a test), then discovery, then the default. Returns (root, name);
    the root is None when discovery did not run.
    """
    explicit = getattr(args, "docs_dir", None) or os.environ.get(DOCS_DIR_ENV)
    if explicit:
        return None, explicit
    return discover_docs_root(start)


def discover_docs_root(start=None):
    """Walk up looking for a candidate root. PER LEVEL: the nearest one wins.

    Two candidates at the SAME level is the shape of a half-migrated project.
    Returning either would validate half of it and print a verdict, so it refuses
    -- naming both -- and the caller exits without a verdict.
    """
    cur = Path(start or os.getcwd()).resolve()
    for p in [cur] + list(cur.parents):
        found = [c for c in DOCS_DIR_CANDIDATES if (p / c).is_dir()]
        if len(found) > 1:
            raise AmbiguousDocsRoot(p, found)
        if found:
            return p, found[0]
    return None, DEFAULT_DOCS_DIR


class AmbiguousDocsRoot(Exception):
    """Two documentation roots side by side: a verdict here would be a guess."""

    def __init__(self, where, found):
        self.where = where
        self.found = found
        super().__init__(
            f"{where} contains more than one documentation root ({', '.join(found)}): "
            "refusing to guess which one this project uses. Pass --docs-dir <name> "
            "to say which, or finish the migration so only one remains.")


# --------------------------------------------------------------------- domains
# One entry per lens of the family. The DATA lives here, in the shared core, so a
# mixed tree is validated identically from every installed distribution: an entry
# point never picks rules, it only declares which portable checks it can run.
#
# The exclusive part of a rule set -- template, mandatory sections, the risk slot
# -- has exactly one owner per document. It is the part that never composes:
# unioning it would demand every domain's ceremony of every document, intersecting
# it would demand nothing. The risk slot is translated per domain, never dropped.
DOMAINS = {
    "code": {
        "risk_section": SECURITY_SECTION,
        "risk_label": "## Security and Threat Model",
        "id_prefix": "F-",
    },
    "knowledge": {
        "risk_section": ("## Sources and Verification",),
        "risk_label": "## Sources and Verification",
        "id_prefix": "K-",
    },
    "marketing": {
        "risk_section": ("## Threat Map / Plan Risks", "## Threat Map"),
        "risk_label": "## Threat Map / Plan Risks",
        "id_prefix": "M-",
    },
}
# Absent everything -- no `default_domain:` line, no `domain:` field -- a project is
# `code`. That is what every project created before this field existed already is,
# so the default is chosen to leave them untouched, not because code is special.
DEFAULT_DOMAIN = "code"

# Portable checks: composable, opt-in per document via `checks:`. An imported check
# may only ADD findings, never relax what the owning domain requires -- monotonic, so
# importing one is safe by construction. Registered by name `<domain>.<check>`.
PORTABLE_CHECKS = {}
# Which check namespaces this distribution actually carries. Set by the entry point;
# a `checks:` entry outside it WARNS visibly rather than passing silently.
_ENTRY_POINT = {"domain": DEFAULT_DOMAIN, "provides": (), "script": "sdlc_check.py"}
# "script" is what generated headers tell the reader to RUN. Declared, never
# derived from sys.argv: the generated bytes must not depend on how the command
# was invoked, or the alignment check would fail on the invocation instead of on
# the content.


# --- distribution profile ----------------------------------------------------
# Each distribution declares what it carries. The battery reads this instead of
# assuming the code overlay, which is what lets one shared battery run in three
# distributions without either failing on files a domain legitimately does not have
# or quietly excusing a domain from doctrine it owes.
#
# REQUIRED_CAPABILITIES is the spine: process discipline that is domain-neutral, so
# no distribution may drop it. A profile missing one fails a shared test -- editing
# your own profile is therefore NOT a way out of the doctrine, only a way to declare
# an overlay you genuinely do not have.
REQUIRED_CAPABILITIES = frozenset({
    "triage",              # Rule Zero, with the router verdict as a declared output
    "write_triggers",      # one event, one destination
    "workstream_registry", # audit/handoff.md as a parallel-safe registry
    "vision_gate",         # DRAFT informs, APPROVED binds, blind check before promotion
    "design_review_gate",  # a design reviewed by somebody other than its author
    "guide_router",        # the mandatory pre-work lookup
    "worktree_hygiene",    # isolate the work
})
# Optional overlays: real capabilities that a domain may legitimately not have.
# Listed here so "this distribution does not claim it" is a visible decision.
OPTIONAL_CAPABILITIES = frozenset({
    "architect_pass",         # does the component already exist? (code overlay)
    "interaction_contract",   # actor-facing surface spec between use cases and solution (code overlay)
    "taxonomy_pass",          # do the categories/topics already exist? (knowledge overlay)
    "comprehension_guides",   # source_kind: code maps of complex components
    "tdd",                    # test-first discipline
    "subagent_dispatch",      # opt-in PLAN_[feature].md execution
    "legacy_narrative_handoff",  # published before the registry format: owes a migration clause
    "question_discipline",    # when a question to the user is legal (elicitation.md);
                              # spine candidate once every sibling's elicitation carries it
})
# `unit_noun` is vocabulary, not structure: the code domain works on a "feature",
# the knowledge domain on a "topic". The shared battery asserts the SHAPE
# (HANDOFF_[<unit>].md) and reads the word from here, so a domain keeps its own
# language without either weakening the assertion or forking the test.
# `design_gate_between` is the pair of SKILL.md headings the design review must sit
# between: after the design exists, before the work is executed. The headings are each
# domain's own wording -- the code overlay runs five phases, the marketing overlay nine
# -- so the battery asserts the ORDER, never a shared phase name.
_PROFILE = {
    "skill_name": "agentic-sdlc",
    "unit_noun": "feature",
    "support_files": (),
    "capabilities": frozenset(),
    "design_gate_between": (),
}


def set_profile(skill_name, support_files=(), capabilities=(), unit_noun="feature",
                design_gate_between=()):
    _PROFILE.update(skill_name=skill_name,
                    unit_noun=unit_noun,
                    support_files=tuple(support_files),
                    capabilities=frozenset(capabilities),
                    design_gate_between=tuple(design_gate_between))


def profile():
    return dict(_PROFILE)


def has_capability(name):
    return name in _PROFILE["capabilities"]


def portable_check(name):
    """Register a portable check. The callable takes (rel, meta, text) and returns
    a list of (severity, message) with severity in {'error', 'warning', 'advisory'}."""
    def register(fn):
        PORTABLE_CHECKS[name] = fn
        return fn
    return register


def set_entry_point(domain, provides=(), script=None):
    """Declare which domain this distribution is and which check namespaces it ships."""
    _ENTRY_POINT["domain"] = domain
    _ENTRY_POINT["provides"] = tuple(provides)
    if script:
        _ENTRY_POINT["script"] = script


def entry_script():
    """The command name a generated header tells the reader to run."""
    return _ENTRY_POINT["script"]


def project_default_domain(root):
    """The project's answer for every artifact that declares no `domain:`.

    Read once from `ai_docs/README.md`'s frontmatter. Project-level ON PURPOSE: a
    per-distribution default would give the same tree two different verdicts
    depending on which lens the agent happened to load."""
    readme = ai_path(root, "README.md")
    if not readme.is_file():
        return DEFAULT_DOMAIN
    meta = load_frontmatter(read_text(readme).splitlines())
    declared = (meta.get("default_domain") or "").strip().strip("'\"").lower()
    return declared if declared in DOMAINS else DEFAULT_DOMAIN


def resolve_domain(meta, default):
    """(domain, declared_but_unknown) for one artifact. The field RECORDS the answer;
    it never invents one, and an unrecognized value is reported, not obeyed."""
    declared = (meta.get("domain") or "").strip().strip("'\"").lower()
    if not declared:
        return default, None
    if declared not in DOMAINS:
        return default, declared
    return declared, None


def declared_checks(meta):
    """The `checks:` list, accepted as `[a, b]` or as a comma-separated string."""
    raw = (meta.get("checks") or "").strip()
    if not raw:
        return []
    return [c.strip().strip("'\"") for c in raw.strip("[]").split(",") if c.strip()]


def run_portable_checks(rel, meta, text, errors, warnings, advisories):
    """Run the checks a document imported. Findings are ADDED to the owning domain's;
    an unavailable check is a visible warning -- never a silent pass."""
    for name in declared_checks(meta):
        namespace = name.split(".", 1)[0]
        if name not in PORTABLE_CHECKS:
            if namespace in _ENTRY_POINT["provides"]:
                warnings.append(f"{rel}: check '{name}' is unknown (no such portable check)")
            else:
                warnings.append(
                    f"{rel}: check '{name}' is not available in this distribution "
                    f"(it ships {', '.join(_ENTRY_POINT['provides']) or 'no checks'}): "
                    "the document was NOT checked against it")
            continue
        for severity, message in PORTABLE_CHECKS[name](rel, meta, text) or []:
            {"error": errors, "warning": warnings}.get(severity, advisories).append(
                f"{rel}: [{name}] {message}")


try:
    sys.stdout.reconfigure(encoding="utf-8", errors="replace")
    sys.stderr.reconfigure(encoding="utf-8", errors="replace")
except Exception:
    pass


# --- orient (SessionStart hook) ---
# Fixed, hard-coded doc set (label, path-relative-to-root). No content- or
# user-derived paths -> no traversal input (P-TM T3); confine_under is
# defense-in-depth. Emitted at session start by the orient subcommand.
def orient_docs():
    d = docs_dir()
    return [
        ("Reading guide (README)", f"{d}/README.md"),
        ("Canonical manifest (INDEX)", f"{d}/INDEX.md"),
        ("Guide router (when-to-consult)", f"{d}/reference/INDEX.md"),
        ("Last session handoff", f"{d}/audit/handoff.md"),
    ]
ORIENT_PER_DOC_CHARS = 6000     # per-doc truncation
ORIENT_MAX_TOTAL_CHARS = 16000  # total ingestion cap (P-TM T2); tunable


# ----------------------------------------------------------------- utilities

def utc_now_iso():
    return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")


def find_project_root(start=None):
    cur = Path(start or os.getcwd()).resolve()
    for p in [cur] + list(cur.parents):
        if (p / docs_dir()).is_dir():
            return p
    return cur


def require_ai_docs(root, command):
    """Fail fast when the docs root is missing: prevents silently creating a second
    documentation root in the wrong working directory."""
    if not ai_path(root).is_dir():
        print(f"[ERROR] {ai_path(root)} not found: refusing to run '{command}' here. "
              "Run agentic-sdlc-init first, or pass --root <project_root>.")
        return False
    return True


def confine_under(base, rel):
    """Fail-closed path confinement: resolve `rel` under `base` and require the
    result to stay inside `base`. Returns None (reject) if `rel` is absolute,
    contains a '..' part, or resolves outside `base` (including an OSError
    during resolution, e.g. an unresolvable/reparse-point path on Windows).
    Single source for path confinement (T2/T3): reused by check_kb_collisions'
    `overrides:` check and cmd_validate's `distilled_from` check, and by the
    new `plan` command's paths/consumes/produces/guides confinement."""
    p = Path(rel)
    if p.is_absolute() or ".." in p.parts:
        return None
    try:
        t = (base / rel).resolve()
        t.relative_to(base.resolve())
        return t
    except (ValueError, OSError):
        return None


def read_text(path):
    # utf-8-sig: strips a leading BOM (files authored on Windows) so the
    # frontmatter '---' on line 0 stays recognizable; reads plain utf-8 otherwise.
    return path.read_text(encoding="utf-8-sig", errors="replace")


def sha256_file(path):
    # CRLF->LF before hashing: a Windows checkout with core.autocrlf=true
    # rewrites snapshot files, and a raw-byte hash would flag every guide
    # [stale] on a fresh clone. Recorded hashes are LF-based, so normalizing
    # maps CRLF copies back to the same digest.
    h = hashlib.sha256()
    h.update(path.read_bytes().replace(b"\r\n", b"\n"))
    return h.hexdigest()


def parse_iso(value):
    if not value:
        return None
    try:
        dt = datetime.fromisoformat(value.strip().replace("Z", "+00:00"))
        if dt.tzinfo is None:
            dt = dt.replace(tzinfo=timezone.utc)
        return dt
    except ValueError:
        return None


def norm_text(s):
    return "\n".join(line.rstrip() for line in s.strip().splitlines())


def load_frontmatter(lines):
    meta = {}
    if not lines or lines[0].strip() != "---":
        return meta
    for line in lines[1:60]:
        if line.strip() == "---":
            break
        m = re.match(r"^([A-Za-z_][\w-]*):\s*(.*)$", line)
        if m:
            meta[m.group(1).strip().lower()] = m.group(2).strip()
    # Legacy Italian keys: accepted, normalized to canonical English (deprecated).
    for legacy, canon in LEGACY_KEYS.items():
        if legacy in meta and canon not in meta:
            meta[canon] = meta[legacy]
    return meta


def is_shadow(path, first_line):
    """A shadow mirror of a devPNT-governed document, not an authoritative ANALYSIS.
    Recognized by filename (SHADOW_*) or by the marker comment on the FIRST line
    (legacy shadows saved under an ANALYSIS_* name)."""
    return path.name.startswith("SHADOW") or first_line.lstrip().startswith("<!-- SHADOW")


def list_analyses(root):
    """Returns [(path, frontmatter, text)] for the ANALYSIS_*.md files (shadows excluded)."""
    sol = ai_path(root, "solutions")
    out = []
    if not sol.is_dir():
        return out
    for p in sorted(sol.glob("ANALYSIS_*.md")):
        text = read_text(p)
        first_line = text.splitlines()[0] if text else ""
        if is_shadow(p, first_line):
            continue
        out.append((p, load_frontmatter(text.splitlines()), text))
    return out


def has_etdd_shadow(root):
    """True if an E-TDD shadow exported from devPNT exists in solutions/.
    In Hybrid mode the approved E-TDD (exported BEFORE implementation) is the
    design authorization that replaces the IN_PROGRESS ANALYSIS."""
    sol = ai_path(root, "solutions")
    if not sol.is_dir():
        return False
    return any("tdd" in p.name.lower() for p in sol.glob("SHADOW_*.md"))


def iter_files(target):
    if target.is_file():
        yield target
        return
    for dirpath, dirnames, filenames in os.walk(target):
        dirnames[:] = [d for d in dirnames if d not in SKIP_DIRS and d != docs_dir() and not d.startswith(".")]
        for name in filenames:
            yield Path(dirpath) / name


# ---------------------------------------------------------------------- git

def git_available(root):
    try:
        r = subprocess.run(["git", "rev-parse", "--is-inside-work-tree"],
                           cwd=str(root), capture_output=True, text=True, timeout=10)
        return r.returncode == 0 and r.stdout.strip() == "true"
    except Exception:
        return False


def git_head(root):
    try:
        r = subprocess.run(["git", "rev-parse", "--short=12", "HEAD"],
                           cwd=str(root), capture_output=True, text=True, timeout=10)
        return r.stdout.strip() if r.returncode == 0 else ""
    except Exception:
        return ""


def git_has_changes(root, rel_path):
    """True if there are tracked/untracked changes under rel_path."""
    try:
        rel = rel_path.replace("\\", "/")
        r = subprocess.run(["git", "status", "--porcelain", "--", rel],
                           cwd=str(root), capture_output=True, text=True, timeout=30)
        return r.returncode == 0 and bool(r.stdout.strip())
    except Exception:
        return False


def git_changed_since(root, ref, rel_path):
    """Files changed (tracked + untracked) under rel_path since ref. None if ref unresolvable."""
    try:
        r = subprocess.run(["git", "diff", "--name-only", ref, "--", rel_path],
                           cwd=str(root), capture_output=True, text=True, timeout=30)
        if r.returncode != 0:
            return None
        changed = [l.strip() for l in r.stdout.splitlines() if l.strip()]
        r2 = subprocess.run(["git", "ls-files", "--others", "--exclude-standard", "--", rel_path],
                            cwd=str(root), capture_output=True, text=True, timeout=30)
        if r2.returncode == 0:
            changed += [l.strip() for l in r2.stdout.splitlines() if l.strip()]
        return sorted(set(changed))
    except Exception:
        return None


# -------------------------------------------------------------------- index

def build_index(root):
    analyses = list_analyses(root)
    # SYNTACTIC predicate, on purpose: the column appears when some analysis WRITES
    # a `domain:` field, not when one resolves to a domain (every analysis does).
    # So the generated file is a function of the tree alone -- identical from every
    # entry point, and byte-identical on every tree that predates the field.
    tagged = any((meta.get("domain") or "").strip() for _, meta, _ in analyses)
    default_domain = project_default_domain(root) if tagged else None
    rows = []
    for p, meta, _ in analyses:
        row = [
            meta.get("id", "?"),
            meta.get("feature", p.stem.replace("ANALYSIS_", "")),
            meta.get("level", ""),
            meta.get("status", "?"),
            meta.get("start_date", ""),
            meta.get("end_date", ""),
            "solutions/" + p.name,
        ]
        if tagged:
            row.insert(2, resolve_domain(meta, default_domain)[0])
        rows.append(tuple(row))
    rows.sort(key=lambda r: r[0])
    header = "| ID | Feature | Level | Status | Started | Finished | Doc |"
    sep = "|---|---|---|---|---|---|---|"
    if tagged:
        header = "| ID | Feature | Domain | Level | Status | Started | Finished | Doc |"
        sep = "|---|---|---|---|---|---|---|---|"
    lines = [INDEX_HEADER,
             "# Feature History (generated)",
             "",
             header,
             sep]
    for r in rows:
        lines.append("| " + " | ".join(r) + " |")
    return "\n".join(lines) + "\n"


# "Status:"/"Stato:" line in the body (with or without ** **), prefix before the description
_STATUS_LINE = re.compile(r"^\**\s*(?:status|stato)\s*\**\s*:\s*\**\s*([A-Za-z][\w-]*)", re.I)
# pure metadata lines to skip when picking the fallback description
_META_LINE = re.compile(r"^\**\s*(date|data|task ref|version|versione|owner|autore|branch|agente|agent|created|creato|updated|aggiornato)\b", re.I)


def extract_doc_meta(path):
    """(title, description, status, supersedes) of a canonical doc.

    Recognizes TWO header conventions: the YAML-lite frontmatter
    (description/status/supersedes/title) and the in-body `**Status:** X`
    line (used by ADRs and legacy docs). As a fallback it derives the title
    from the first '# H1' and the description from the first prose line,
    skipping metadata lines.
    """
    text = read_text(path)
    lines = text.splitlines()
    meta = load_frontmatter(lines)
    body = lines
    if lines and lines[0].strip() == "---":
        for i in range(1, min(len(lines), 60)):
            if lines[i].strip() == "---":
                body = lines[i + 1:]
                break

    title = meta.get("title", "")
    if not title:
        for line in body:
            m = re.match(r"^#\s+(.*)$", line)
            if m:
                title = m.group(1).strip()
                break
    title = title or path.stem

    status = meta.get("status", "").upper()
    if not status:
        for line in body[:25]:
            m = _STATUS_LINE.match(line.strip())
            if m:
                status = m.group(1).upper()
                break

    desc = meta.get("description", "")
    if not desc:
        in_comment = False
        for line in body:
            s = line.strip()
            # track HTML-comment state across lines: skipping only the OPENING
            # line made line 2 of a multi-line comment the manifest description
            # (the shipped vision template opens with a 3-line comment, so the
            # most-read row of the manifest read '... -->')
            if in_comment:
                if "-->" in s:
                    in_comment = False
                    s = s.split("-->", 1)[1].strip()
                    if not s:
                        continue
                else:
                    continue
            elif s.startswith("<!--"):
                if "-->" not in s:
                    in_comment = True
                    continue
                s = s.split("-->", 1)[1].strip()
                if not s:
                    continue
            # a table row or a bare bullet is not a description: the manifest is
            # the first thing an agent reads to orient, and '| Milestone | ... |'
            # in that column is a row carrying no information
            if (not s or s.startswith("#") or s.startswith("|") or s.startswith("---")
                    or re.match(r"^[-*+]\s", s) or _META_LINE.match(s)):
                continue
            if s.startswith(">"):
                s = s.lstrip(">").strip()
            m = _STATUS_LINE.match(s)
            if m:
                # "Status: X — description": keep the part after the status; if empty, skip
                rest = s[m.end():].strip(" *—–-:.")
                if not rest:
                    continue
                s = rest
            if s:
                desc = s
                break
    desc = re.sub(r"\s+", " ", desc).strip()
    if len(desc) > 160:
        desc = desc[:157].rstrip() + "..."
    return title, desc, status, meta.get("supersedes", "").strip()


def list_canonical_docs(root):
    """[(rel_to_ai_docs, path, (title, desc, status, supersedes))] for canonical docs."""
    ai = ai_path(root)
    out = []
    for d in MANIFEST_DIRS:
        base = ai / d
        if not base.is_dir():
            continue
        for p in sorted(base.rglob("*.md")):
            rel_parts = p.relative_to(base).parts
            if any(part.startswith(".") for part in rel_parts[:-1]):
                continue  # dot-subdirs (e.g. reference/.sources/) are never canonical
            if p.name in GENERATED_DOCS or p.name == "README.md":
                continue
            out.append((p.relative_to(ai).as_posix(), p, extract_doc_meta(p)))
    return out


def build_manifest(root):
    docs = list_canonical_docs(root)
    lines = [manifest_header(),
             f"# `{docs_dir()}/` document index (generated)",
             "",
             "Complete manifest of the canonical documents. For the reading priority",
             "(must-reads) see the hand-curated `README.md`. The ANALYSIS history is in",
             "`strategic/features_history.md`. `audit/` and `solutions/` are discovery-by-grep,",
             "not manifested here."]
    by_dir = {}
    for rel, _, meta in docs:
        by_dir.setdefault(rel.split("/", 1)[0], []).append((rel, meta))
    for top in MANIFEST_DIRS:
        rows = by_dir.get(top)
        if not rows:
            continue
        lines += ["", f"## {top}/", "",
                  "| Document | Status | Description |", "|---|---|---|"]
        for rel, (title, desc, status, _sup) in rows:
            d = (desc or title).replace("|", "\\|")
            lines.append(f"| `{rel}` | {status or '-'} | {d} |")
    return "\n".join(lines).rstrip() + "\n"


# ------------------------------------------------- workstream registry (F-028)
# audit/handoff.md is GENERATED from one source file per open workstream, so two
# writers working two workstreams touch two different files. Row-per-workstream
# alone was not enough -- F-019 had that and the file still conflicted twice,
# because a file-global `Date:` header defeats row-level ownership. The header is
# now DERIVED, so no writer touches it.

REGISTRY_COLUMNS = ("Workstream", "Level", "Branch", "Status", "Since",
                    "Next step", "Details")
REGISTRY_KEYS = ("workstream", "level", "branch", "status", "since", "next")
REGISTRY_CAP = 20               # a signal, never a truncation (see cmd_index)
PROJECT_NOTES = "project_notes.md"   # NOT handoff_notes.md: the HANDOFF_*.md
# glob is case-insensitive on Windows, and that name would be collected as a
# source. The trap is real; the name is the fix.


def registry_header():
    return (f"<!-- GENERATED by {entry_script()} index - do not edit by hand. "
            f"Source of truth: the HANDOFF_*.md files in {docs_dir()}/audit/. -->")


def list_workstreams(root):
    """[(path, meta)] for audit/HANDOFF_*.md carrying registry frontmatter.

    Opt-in by the presence of `workstream:` -- a project whose handoff is still
    hand-written has no sources, so nothing generates and nothing errors (the
    F-019 migration lesson). Sorted by workstream id: the alignment check is a
    byte comparison, and glob order differs across filesystems."""
    aud = ai_path(root, "audit")
    out = []
    if not aud.is_dir():
        return out
    for p in sorted(aud.glob("HANDOFF_*.md")):
        meta = load_frontmatter(read_text(p).splitlines())
        if str(meta.get("workstream") or "").strip():
            out.append((p, meta))
    out.sort(key=lambda pm: (str(pm[1].get("workstream")).strip().lower(), pm[0].name))
    return out


def _registry_cell(value):
    text = str(value if value is not None else "").strip()
    return text.replace("|", "\\|") or "-"


def build_registry(root):
    """The generated registry, or "" when there is no source to build it from.

    Deterministic by construction: sorted rows, and a `Date:` taken from the
    newest `updated:` VALUE written inside the sources -- never a filesystem
    timestamp. Git does not preserve mtimes, so an mtime-derived header would
    regenerate differently in every fresh clone and the alignment check would
    fire on a tree nobody touched."""
    rows = list_workstreams(root)
    if not rows:
        return ""
    stamp = max(str(m.get("updated") or m.get("since") or "").strip() or "0000-00-00"
                for _p, m in rows)
    lines = ["# Handoff — workstream registry",
             f"Date: {stamp} (UTC)",
             "",
             registry_header(),
             "",
             "| " + " | ".join(REGISTRY_COLUMNS) + " |",
             "|" + "---|" * len(REGISTRY_COLUMNS)]
    for p, meta in rows:
        extra = str(meta.get("details") or "").strip()
        details = f"{p.name} · {extra}" if extra else p.name
        cells = [_registry_cell(meta.get(k)) for k in REGISTRY_KEYS]
        lines.append("| " + " | ".join(cells) + f" | {_registry_cell(details)} |")
    notes = ai_path(root, "audit", PROJECT_NOTES)
    if notes.is_file():
        body = read_text(notes).strip()
        if body:
            lines += ["", "## Project-wide notes", "", body]
    return "\n".join(lines) + "\n"


def parse_registry_rows(text):
    """Workstream ids in a registry table, hand-written or generated."""
    ids = []
    for line in text.splitlines():
        s = line.strip()
        if not s.startswith("|") or set(s) <= {"|", "-", " ", ":"}:
            continue
        first = s.strip("|").split("|")[0].strip()
        if first and first.lower() != "workstream":
            ids.append(first)
    return ids


def registry_conversion_blockers(root):
    """What stops `index` from writing over a hand-written handoff.md.

    The mixed state is the trap this exists for: converting one row at a time
    leaves a project at one source file and five hand-written rows, and
    regenerating from the one source DELETES the other five -- silently, in the
    file whose whole purpose is not losing them. So conversion is per project.
    Empty list = writing is safe."""
    hand = ai_path(root, "audit", "handoff.md")
    if not hand.is_file():
        return []
    text = read_text(hand)
    # "Already ours" = written by ANY family entry point: the header is WRITTEN
    # with entry_script(), so recognition must not hard-code one distribution's
    # script name (a registry generated by mkt_check.py is just as generated).
    if re.search(r"GENERATED by \S+ index - do not edit by hand", text):
        return []
    blockers = []
    known = {str(m.get("workstream")).strip() for _p, m in list_workstreams(root)}
    orphans = [r for r in parse_registry_rows(text) if r not in known]
    if orphans:
        blockers.append("rows no HANDOFF_*.md accounts for: " + ", ".join(orphans))
    # Everything else in the file must have a home too, or it is lost on write:
    # a pre-1.17 narrative handoff carries no table at all, so orphan rows alone
    # would not notice it.
    notes_ok = ai_path(root, "audit", PROJECT_NOTES).is_file()
    leftovers, in_notes = [], False
    for line in text.splitlines():
        s = line.strip()
        if not s or s.startswith("|") or s.startswith("<!--"):
            continue
        if s.startswith("# ") or re.match(r"^(?:Date|Data):", s):
            continue
        if re.match(r"^##\s+Project-wide notes\s*$", s):
            in_notes = True
            continue
        if s.startswith("## "):
            in_notes = False
        if in_notes and notes_ok:
            continue
        leftovers.append(s)
    if leftovers:
        blockers.append("content outside the table with nowhere to go (%d line(s), first: %r) "
                        "-- project-wide notes belong in audit/%s"
                        % (len(leftovers), leftovers[0][:60], PROJECT_NOTES))
    return blockers


def list_guides(root):
    """[(rel_to_ai_docs, path, meta, text)] for ai_docs/reference/GUIDE_*.md."""
    ref = ai_path(root, "reference")
    out = []
    if not ref.is_dir():
        return out
    for p in sorted(ref.glob("GUIDE_*.md")):
        text = read_text(p)
        out.append((p.relative_to(ai_path(root)).as_posix(), p,
                    load_frontmatter(text.splitlines()), text))
    return out


def check_kb_collisions(root, project_guides, errors, warnings):
    """Cross-root awareness (unit 2): project-wins precedence, declared via 'overrides:'."""
    kb_root = DEFAULT_KB_ROOT
    # The agent-global KB is client-agnostic and shared across lenses: it keeps its
    # own fixed layout and NEVER follows a project's docs-root name (TS16).
    kb_ref = (kb_root / "ai_docs" / "reference")
    try:
        if root.resolve() == kb_root.resolve():
            return  # validating the KB itself: no self-comparison
    except OSError:
        return
    if not kb_ref.is_dir():
        return  # no KB on this machine: zero behavior change
    kb_names = {p.name for _, p, _, _ in list_guides(kb_root)}
    for rel, p, meta, _ in project_guides:
        ov = (meta.get("overrides") or "").strip()
        if ov:
            # T6: untrusted cross-root pointer — distilled_from parity, fail closed
            target = confine_under(kb_ref, ov)
            if target is None:
                errors.append(f"{rel}: overrides '{ov}' is absolute, contains '..', or escapes the KB "
                              "reference dir — rejected (fail closed)")
                continue
            if not target.is_file():
                warnings.append(f"{rel}: overrides target '{ov}' not found in KB ({kb_ref})")
        if p.name in kb_names and ov != p.name:
            warnings.append(f"{rel}: undeclared collision with KB guide '{p.name}' (project wins) — declare overrides: {p.name}")


def build_guide_index(root):
    lines = [guide_index_header(),
             "# Operative guides (generated router)",
             "",
             "One row per guide. `description` is the when-to-consult line; provenance",
             "shows what the guide was distilled from. Freshness: run `sdlc_check.py stale`.",
             "",
             "| Guide | Status | When to consult | Source | Source version |",
             "|---|---|---|---|---|"]
    for rel, p, meta, _ in list_guides(root):
        lines.append("| `{}` | {} | {} | {} | {} |".format(
            p.name, meta.get("status", "-") or "-",
            (meta.get("description", "") or "-").replace("|", "\\|"),
            (meta.get("source", "") or "-").replace("|", "\\|"),
            meta.get("source_version", "") or "-"))
    return "\n".join(lines) + "\n"


def cmd_index(root):
    if not require_ai_docs(root, "index"):
        return 1
    hist = ai_path(root, "strategic", "features_history.md")
    hist.parent.mkdir(parents=True, exist_ok=True)
    hist.write_text(build_index(root), encoding="utf-8")
    print(f"[ok] ANALYSIS index regenerated: {hist}")
    # INDEX.md only if canonical docs exist: no empty manifest on minimal projects
    if list_canonical_docs(root):
        manifest = ai_path(root, "INDEX.md")
        manifest.write_text(build_manifest(root), encoding="utf-8")
        print(f"[ok] document manifest regenerated: {manifest}")
    else:
        print("[info] no canonical documents: INDEX.md not generated")
    guides = list_guides(root)
    gidx = ai_path(root, "reference", "INDEX.md")
    if guides:
        gidx.write_text(build_guide_index(root), encoding="utf-8")
        print(f"[ok] guide router regenerated: {gidx}")
    else:
        # An EMPTY router still gets written: Rule Zero makes reading it a
        # mandatory, declared step, and `no match` may not be faked. Without the
        # stub, the required verdict is unsatisfiable on every new project --
        # and a rule that cannot be obeyed on first contact gets discarded.
        gidx.parent.mkdir(parents=True, exist_ok=True)
        gidx.write_text(guide_index_header() + "\n# Operative guides (generated router)\n\n"
                        "No guides in this project yet. This file exists so the Rule Zero "
                        "router lookup has something to read: the honest verdict here is "
                        "`router: no match`.\n\n"
                        "A guide is written when the user hands over indications to follow "
                        "(`source_kind: document`), or when a high-complexity component needs "
                        "a comprehension map (`source_kind: code`) -- see `guides.md`.\n",
                        encoding="utf-8")
        print(f"[ok] guide router regenerated (empty stub): {gidx}")
    return max(rc_registry(root), 0)


def rc_registry(root):
    """Write the generated workstream registry, or refuse and say why (F-028)."""
    ws = list_workstreams(root)
    if not ws:
        return 0                # no sources: a hand-written handoff is untouched
    blockers = registry_conversion_blockers(root)
    hand = ai_path(root, "audit", "handoff.md")
    if blockers:
        print(f"[ERROR] {docs_dir()}/audit/handoff.md NOT regenerated -- it still holds "
              "state no source accounts for:")
        for b in blockers:
            print(f"          - {b}")
        print("          Convert the whole registry at once (templates.md): converting "
              "one row at a time is the state that loses the others.")
        return 1
    hand.parent.mkdir(parents=True, exist_ok=True)
    hand.write_text(build_registry(root), encoding="utf-8")
    print(f"[ok] workstream registry regenerated: {hand}")
    if len(ws) > REGISTRY_CAP:
        print(f"[warn] {len(ws)} open workstreams: the registry is meant to stay under "
              f"{REGISTRY_CAP}. Nothing was truncated -- closing one is the fix.")
    return 0


# ----------------------------------------------------------------- validate

def has_section(text, aliases):
    return any(a in text for a in aliases)


def section_body(text, aliases):
    """The body of the first matching `## ` section, or None if no alias is present.

    Portable checks read a section rather than the whole document, so a phrase that
    happens to appear elsewhere cannot satisfy a check about this section."""
    lines = text.splitlines()
    for i, line in enumerate(lines):
        if not any(line.startswith(a) for a in aliases):
            continue
        body = []
        for nxt in lines[i + 1:]:
            if nxt.startswith("## "):
                break
            body.append(nxt)
        return "\n".join(body).strip()
    return None


def design_review_due(meta):
    """True when an L3 ANALYSIS owes a design-review row (review.md moment 1):
    implementation has started or finished, and it began on/after the gate
    shipped. PLANNED is exempt -- the review is due at the END of Phase 3, so an
    analysis still being drafted is not late."""
    if meta.get("level", "").upper() != "L3":
        return False
    if meta.get("status") not in ("IN_PROGRESS", "COMPLETED"):
        return False
    started = parse_iso((meta.get("start_date") or "").strip().strip("'\""))
    return started is not None and started >= parse_iso(DESIGN_REVIEW_EPOCH)


def review_logged(root, analysis_name):
    """True when REVIEW_LOG.md carries a design-moment row naming this ANALYSIS.
    The filename matches anywhere in the row (loose on purpose: a freshness
    signal must not turn a formatting slip into a false 'you skipped the
    review'), but the moment is read from the `tier` COLUMN -- the schema
    reserves it for exactly this, and matching 'design' anywhere in the row let
    a CLOSURE row saying 'conformance to the design' satisfy the check."""
    log = root / review_log_rel()
    if not log.is_file():
        return False
    stem = analysis_name[:-3] if analysis_name.endswith(".md") else analysis_name
    # match the filename on a word boundary: a plain substring lets a longer
    # sibling (ANALYSIS_vision_clarity) satisfy a shorter one (ANALYSIS_vision)
    name_re = re.compile(r"(?<![\w-])" + re.escape(stem) + r"(?![\w-])")
    tier_idx = None
    for line in read_text(log).splitlines():
        line = line.strip()
        if not line.startswith("|"):
            continue
        cells = [c.strip() for c in line.strip("|").split("|")]
        if tier_idx is None:
            lowered = [c.lower() for c in cells]
            if "tier" in lowered:          # header found: trust it over position
                tier_idx = lowered.index("tier")
                continue
        if not name_re.search(line):
            continue
        # schema: | date | doc_key | tier | reviewer | raised | real | verdict | rounds |
        idx = tier_idx if tier_idx is not None else 2
        if len(cells) > idx and re.match(r"design\b", cells[idx], re.I):
            return True
    return False


def has_ledger_heading(text):
    """True when a REAL '## Capability Ledger' heading exists: fenced code
    blocks are removed first, then HTML comments. An unterminated '<!--' only
    opens a comment at the start of a line -- nuking to EOF on an inline
    mention (or an unclosed example inside a fence) made a document that
    HAS its ledger get told it has none."""
    stripped = re.sub(r"^(```|~~~).*?^\1", "", text, flags=re.M | re.S)
    stripped = re.sub(r"<!--.*?-->", "", stripped, flags=re.S)
    stripped = re.sub(r"^[ \t]*<!--(?!.*?-->).*\Z", "", stripped, flags=re.M | re.S)
    return bool(re.search(r"^##[ \t]+Capability Ledger[ \t]*$", stripped, re.M))


def ledger_due(meta):
    """True when an ANALYSIS owes a '## Capability Ledger' (architect.md):
    an L3 started on/after the day the pass shipped. Grandfathered by
    start_date ALONE -- deliberately NOT by status: closure flips the ANALYSIS
    to COMPLETED before `check` runs (SKILL.md phase 5), so a status filter
    would silence the backstop at the only moment the process mandates the
    validator. A malformed/absent start_date is not due (fail-open: cmd_validate
    already errors on a missing one, and guessing an epoch from garbage would
    nag projects the pass never reached)."""
    if meta.get("level", "").upper() != "L3":
        return False
    if meta.get("status") == "CANCELLED":
        return False   # abandoned work has no legitimate way to satisfy this
    started = parse_iso((meta.get("start_date") or "").strip().strip("'\""))
    return started is not None and started >= parse_iso(ARCHITECT_PASS_EPOCH)


MAP_SECTION_RE = re.compile(r"^#{2,3}[ \t]+Component Map\b.*?$(.*?)(?=^#{1,3}[ \t]+\S|\Z)",
                            re.M | re.S | re.I)


def map_where_refs(arch_text):
    """Normalized, symbol-stripped paths from the Component Map's 'Where' column
    ONLY -- never from the whole document. Harvesting the whole file let the
    canonical template's own '## Directory Structure' backticks satisfy the check
    and silently disable it on every project that fills that section in.
    Returns None when the document has no Component Map at all."""
    m = MAP_SECTION_RE.search(arch_text)
    if not m:
        return None
    refs, where_idx = [], None
    for ln in m.group(1).splitlines():
        ln = ln.strip()
        if not ln.startswith("|"):
            continue
        cells = [c.strip() for c in ln.strip("|").split("|")]
        if len(cells) < 2 or not cells[0] or set(cells[0]) <= {"-", ":"}:
            continue
        if where_idx is None:
            lowered = [c.lower() for c in cells]
            if "where" not in lowered:
                return []          # no Where column: nothing is mapped
            where_idx = lowered.index("where")
            continue
        if where_idx < len(cells):
            for _ref, path_part, _sym in _map_refs(cells[where_idx]):
                refs.append(path_part.lstrip("./").strip("/"))
    return refs


def _map_refs(where):
    """Backticked refs in a 'Where' cell that are file paths: they contain a
    separator, or end in a KNOWN source-file suffix. Windows separators are
    normalized. Everything else in that cell is prose and must stay silent --
    a false 'the map is rotting' teaches readers to ignore the output, which is
    worse than the rot. `app.core`, `OrderStore.save` and `1.18.0` are prose."""
    out = []
    for ref in re.findall(r"`([^`]+)`", where):
        path_part, _, symbol = ref.partition("#")
        path_part = path_part.replace("\\", "/").strip()
        if not path_part or "://" in path_part:
            continue  # a URL is not a repo path
        # A slash-less token counts only if it looks like a FILENAME. The one
        # real false-positive class is `Next.js` / `Node.js` / `Vue.js`: a
        # CamelCase stem with a `.js` tail is a framework name, not a file.
        # The exclusion is scoped to that suffix ON PURPOSE -- a blanket
        # CamelCase rule would silence `App.tsx`, `Program.cs`, `Main.java`,
        # which are exactly what React/C#/Java projects put in a Where cell.
        stem, _, suffix = path_part.rsplit("/", 1)[-1].rpartition(".")
        framework_name = (suffix.lower() == "js"
                          and bool(re.fullmatch(r"[A-Z][a-z0-9]+(?:[A-Z][a-z0-9]*)*", stem)))
        looks_like_path = "/" in path_part or (
            bool(re.search(r"\.(" + "|".join(FILE_SUFFIXES) + r")$", path_part, re.I))
            and not framework_name)
        if looks_like_path:
            out.append((ref, path_part, symbol.strip()))
    return out


def check_component_map(root, text, advisories):
    """Anti-rot for the '## Component Map' of strategic/architecture.md
    (architect.md): every path-shaped backticked ref in the 'Where' column must
    still resolve on disk, and its '#symbol' must still appear as a whole word
    in a matched file. This is the map's equivalent of the guides' source_hash.
    ADVISORY: a freshness signal, never a gate -- not even under --strict (the
    accepted ceremony budget was a warning, not a blocked pipeline)."""
    m = MAP_SECTION_RE.search(text)   # one regex for both checks: they cannot drift
    if not m:
        return
    rows = [ln.strip() for ln in m.group(1).splitlines() if ln.strip().startswith("|")]
    where_idx, header_cells, checked, data_rows, ragged = None, 0, 0, 0, 0
    for row in rows:
        cells = [c.strip() for c in row.strip("|").split("|")]
        if len(cells) < 2 or not cells[0] or set(cells[0]) <= {"-", ":"}:
            continue
        if where_idx is None:  # the first non-separator row is the header
            lowered = [c.lower() for c in cells]
            if "where" not in lowered:
                advisories.append("strategic/architecture.md: Component Map has no 'Where' "
                                  "column in its header -- the anti-rot check cannot run; "
                                  "give the table a Where column of `path/to/file#Symbol` refs")
                return
            where_idx, header_cells = lowered.index("where"), len(cells)
            continue
        if all(c in ("", "...", "…") for c in cells):
            continue                      # untouched template placeholder row
        data_rows += 1
        if len(cells) != header_cells:    # ragged: never silently unchecked
            ragged += 1
            continue
        component, where = cells[0], cells[where_idx]
        for ref, path_part, symbol in _map_refs(where):
            checked += 1
            if confine_under(root, re.sub(r"[*?\[\]]", "x", path_part)) is None:
                advisories.append(f"strategic/architecture.md: Component Map row '{component}': "
                                  f"ref '{ref}' escapes the project root: rejected")
                continue
            target = root / path_part
            try:
                if target.exists():          # literal first: `app/[id]/page.tsx` is a real path
                    matches = [target]
                elif any(c in path_part for c in "*?["):
                    matches = list(root.glob(path_part))
                else:
                    matches = []
            except (ValueError, OSError):
                matches = []
            if not matches:
                advisories.append(f"strategic/architecture.md: Component Map row '{component}': "
                                  f"'{path_part}' no longer exists -- the map is rotting, "
                                  "update the row or drop it")
                continue
            files = [f for f in matches if f.is_file()]
            if symbol and files:
                word = re.compile(r"(?<![A-Za-z0-9_])" + re.escape(symbol) + r"(?![A-Za-z0-9_])")
                if not any(word.search(read_text(f)) for f in files):
                    advisories.append(f"strategic/architecture.md: Component Map row '{component}': "
                                      f"symbol '{symbol}' not found in '{path_part}' -- renamed or "
                                      "removed, update the row")
    if ragged:
        advisories.append(f"strategic/architecture.md: Component Map has {ragged} row(s) whose "
                          "column count differs from the header -- unchecked; an escaped '|' in "
                          "a cell shifts the columns")
    if data_rows and not checked and not ragged:
        # only once the map claims real components: a freshly seeded project
        # carries the template placeholder and must NOT be nagged on day zero
        advisories.append("strategic/architecture.md: Component Map has rows but no checkable "
                          "path in its 'Where' column -- the anti-rot check is inert; write refs "
                          "as `path/to/file#Symbol`")


def cmd_validate(root, strict=False, hybrid=False):
    # advisories: architect-pass freshness signals. Reported, never escalated by
    # --strict -- the accepted ceremony budget (project_vision.md "no ceremony
    # ratchet") was a warning, and a warning that reddens CI is a gate.
    errors, warnings, advisories = [], [], []
    ai = ai_path(root)
    if not ai.is_dir():
        if strict:
            print(f"[ERROR] {ai} does not exist: nothing to validate. In --strict mode this "
                  "fails so a wrong working directory cannot produce a green pipeline.")
            return 1
        print(f"[info] {ai} does not exist: nothing to validate (project without SDLC docs).")
        return 0

    # Vision: presence and declared state
    for name in VISION_FILES:
        f = ai / "vision" / name
        if not f.is_file():
            warnings.append(f"vision/{name} missing")
            continue
        head = "\n".join(read_text(f).splitlines()[:12])
        m = re.search(r"(?:Status|Stato):\s*(DRAFT|APPROVED)", head)
        if not m:
            errors.append(f"vision/{name}: missing 'Status: DRAFT|APPROVED' in the first lines")
        elif m.group(1) == "DRAFT":
            # advisory, not a warning: bootstrap MANDATES DRAFT, so a warning here
            # makes `validate --strict` red on every freshly bootstrapped project
            # until a human runs the blind check -- and teams delete the CI step
            # rather than block on it. DRAFT is a state, not a defect.
            advisories.append(f"vision/{name} is DRAFT: not a gating authority, "
                              "have the user validate it")

    # ANALYSIS: frontmatter and mandatory sections
    # Ids are unique WITHIN a domain (prefixes F-/K-/M- keep them apart in practice):
    # two lenses over one tree must not collide on a number neither of them chose.
    seen_ids = {}
    default_domain = project_default_domain(root)
    analyses = list_analyses(root)
    for p, meta, text in analyses:
        rel = "solutions/" + p.name
        if not meta:
            errors.append(f"{rel}: frontmatter missing")
            continue
        domain, unknown_domain = resolve_domain(meta, default_domain)
        if unknown_domain:
            warnings.append(f"{rel}: domain '{unknown_domain}' not recognized "
                            f"({'/'.join(sorted(DOMAINS))}): validated as '{domain}'")
        fid = meta.get("id")
        if not fid:
            errors.append(f"{rel}: 'id' field missing")
        elif (domain, fid) in seen_ids:
            errors.append(f"{rel}: id '{fid}' duplicated (already used in {seen_ids[(domain, fid)]})")
        else:
            seen_ids[(domain, fid)] = rel
        status = meta.get("status", "")
        if status not in VALID_STATES:
            errors.append(f"{rel}: status '{status}' not valid ({'/'.join(sorted(VALID_STATES))})")
        if not meta.get("start_date"):
            errors.append(f"{rel}: 'start_date' missing")
        if status == "COMPLETED" and not meta.get("end_date"):
            errors.append(f"{rel}: COMPLETED without 'end_date'")
        level = meta.get("level")
        if level and level.upper() not in VALID_LEVELS:
            warnings.append(f"{rel}: level '{level}' not recognized ({'/'.join(sorted(VALID_LEVELS))})")
        # The risk slot is translated per domain, never dropped: whichever lens owns
        # the document, it owes ITS account of what could go wrong.
        rules = DOMAINS[domain]
        if not has_section(text, rules["risk_section"]):
            errors.append(f"{rel}: section '{rules['risk_label']}' missing (mandatory)")
        for en, it in ANALYSIS_SECTIONS:
            if not has_section(text, (en, it)):
                warnings.append(f"{rel}: section '{en}' missing")
        run_portable_checks(rel, meta, text, errors, warnings, advisories)
        if not level and (parse_iso((meta.get("start_date") or "").strip().strip("'\"")) or
                          parse_iso("1970-01-01")) >= parse_iso(ARCHITECT_PASS_EPOCH):
            # advisory + epoch-gated, exactly like the check it guards: a warning
            # here would redden --strict CI on every pre-1.18 analysis that never
            # carried the optional field. (Same defect the advisories bucket was
            # invented to prevent -- reintroduced once, caught in review.)
            advisories.append(f"{rel}: 'level' missing (L1/L2/L3/Spike) -- risk-proportional "
                              "checks cannot apply, and dropping the line is cheaper than "
                              "doing the work it triggers")
        # comment-stripped, anchored: a '<!-- TODO: the ## Capability Ledger -->'
        # must not read as the section being present
        # Hybrid: the design lives in devPNT and its §4.5 gate owns this slot
        # (SKILL.md ownership matrix: "run ONE of them, never both"), and its log
        # rows are keyed on e_isp_/e_tdd_ doc_keys, not on this filename -- so
        # firing here would be a permanent, unfixable false positive.
        if not hybrid and design_review_due(meta) and not review_logged(root, p.name):
            advisories.append(f"{rel}: L3 in implementation with no design-review row in "
                              f"{review_log_rel()} -- the design was reviewed by nobody but its "
                              "author before code was written (review.md moment 1)")
        if ledger_due(meta) and not has_ledger_heading(text):
            advisories.append(f"{rel}: L3 without '## Capability Ledger' -- the architect pass "
                              "left no record (architect.md); run it before the Impact")

    # Generated index aligned
    hist = ai / "strategic" / "features_history.md"
    if analyses:
        if not hist.is_file():
            errors.append("strategic/features_history.md missing: run 'sdlc_check.py index'")
        elif norm_text(read_text(hist)) != norm_text(build_index(root)):
            errors.append("strategic/features_history.md not aligned with the ANALYSIS files: run 'sdlc_check.py index'")

    # Canonical document manifest aligned (Poka-Yoke: unindexed file = dirty closure)
    docs = list_canonical_docs(root)
    manifest = ai / "INDEX.md"
    if docs:
        if not manifest.is_file():
            errors.append(f"{docs_dir()}/INDEX.md missing: run 'sdlc_check.py index'")
        elif norm_text(read_text(manifest)) != norm_text(build_manifest(root)):
            errors.append(f"{docs_dir()}/INDEX.md not aligned with the canonical documents: run 'sdlc_check.py index'")

    # Canonical document lifecycle: declared status + supersedes coherence
    canon_status = {rel: meta[2] for rel, _, meta in docs}
    for rel, _, (title, desc, status, supersedes) in docs:
        if not status:
            warnings.append(f"{rel}: missing 'status:' in the header (CURRENT/SUPERSEDED/DRAFT/DEPRECATED)")
        elif status not in CANONICAL_STATES:
            warnings.append(f"{rel}: status '{status}' not recognized ({'/'.join(sorted(CANONICAL_STATES))})")
        if supersedes:
            base = os.path.basename(supersedes)
            for other, ost in canon_status.items():
                if (other == supersedes or other.endswith("/" + supersedes)
                        or os.path.basename(other) == base) and ost == "CURRENT":
                    warnings.append(f"{other}: still CURRENT but superseded by {rel} (set status: SUPERSEDED)")

    # Component Map anti-rot (architect.md): rows must still resolve on disk
    arch = ai / "strategic" / "architecture.md"
    if arch.is_file():
        arch_text = read_text(arch)
        check_component_map(root, arch_text, advisories)
        # the missing half of the loop: `mark` asserts an area was read closely
        # enough to name its capability owners, and nothing verified that claim.
        # An ANALYZED area with no map row is how the brownfield guard is
        # disarmed -- the area looks read, so the map's silence becomes groundable.
        _, _, plan_rows = parse_audit_plan(root)
        mapped = map_where_refs(arch_text)
        if plan_rows and mapped is not None:
            for prow in plan_rows:
                if prow["status"] != "ANALYZED":
                    continue
                if confine_under(root, prow["path"]) is None:
                    continue                     # stale already rejects these
                if re.search(r"owns no component", prow.get("note", ""), re.I):
                    continue                     # declared, not forgotten: the opt-out
                area = prow["path"].replace("\\", "/").strip("/")
                if area.startswith("./"):
                    area = area[2:]
                if area in ("", "."):
                    continue                     # the whole root: every row is inside it
                if not (root / area).exists():
                    continue                     # gone from disk: not a mapping gap
                if not any(mp == area or mp.startswith(area + "/") for mp in mapped):
                    advisories.append(
                        f"strategic/architecture.md: '{prow['path']}' is ANALYZED in the audit "
                        "plan but owns no Component Map row -- marking asserts the area was read "
                        "closely enough to name what it owns, and the map's silence there is now "
                        "groundable for a MISSING verdict (architect.md). If it genuinely owns no "
                        "component, say so in the audit plan's Notes column: 'owns no component'")

    # Guide checks (ai_docs/reference/GUIDE_*.md): structure only — freshness is stale's job
    guides = list_guides(root)
    for rel, p, meta, text in guides:
        missing = [k for k in GUIDE_PROVENANCE_KEYS if not meta.get(k)]
        if missing:
            warnings.append(f"{rel}: guide missing provenance key(s): {', '.join(missing)}")
        # (b) per-section fidelity markers: every '## ' section body must carry a marker
        body = text.split("---", 2)[-1]
        sections = re.split(r"^##\s+", body, flags=re.M)[1:]
        unmarked = [s.splitlines()[0].strip() for s in sections if not GUIDE_MARKER_RE.search(s)]
        if unmarked:
            warnings.append(f"{rel}: section(s) without [source: ...] / [not covered by source] marker: "
                            + "; ".join(unmarked[:5]))
        # (c) distilled_from confinement — fail closed (P-TM T6, distilled_from vector)
        df = meta.get("distilled_from", "")
        if df and confine_under(root, df) is None:
            errors.append(f"{rel}: distilled_from '{df}' is absolute, contains '..', or resolves "
                          "outside the project root: rejected")
    check_kb_collisions(root, guides, errors, warnings)
    # guide-router alignment (mirror of the root-manifest check)
    gidx = ai_path(root, "reference", "INDEX.md")
    if not gidx.is_file() and not guides:
        # zero guides: the stub is a convenience for the mandatory Rule Zero read
        advisories.append(f"{docs_dir()}/reference/INDEX.md missing: Rule Zero requires reading the "
                          "guide router and forbids faking its verdict, so the router exists "
                          "even with zero guides -- run 'sdlc_check.py index'")
    if guides:
        if not gidx.is_file():
            # guides EXIST and the router does not: the agent's mandatory lookup
            # finds nothing and legally declares 'absent', so the guide that
            # governs the work is never consulted. An absent router must not be
            # graded below a merely stale one.
            errors.append(f"{docs_dir()}/reference/INDEX.md missing while GUIDE_*.md files exist: "
                          "the router is the only thing that routes work to them -- "
                          "run 'sdlc_check.py index'")
        elif norm_text(read_text(gidx)) != norm_text(build_guide_index(root)):
            errors.append(f"{docs_dir()}/reference/INDEX.md not aligned with the guides: run 'sdlc_check.py index'")

    # Handoff: alignment with its sources (F-028), then header and freshness
    hand = ai / "audit" / "handoff.md"
    workstreams = list_workstreams(root)
    if workstreams:
        if not hand.is_file():
            errors.append(f"{docs_dir()}/audit/handoff.md missing while HANDOFF_*.md sources "
                          "exist: the registry is the only place a cold agent sees the open "
                          "workstreams -- run 'sdlc_check.py index'")
        elif norm_text(read_text(hand)) != norm_text(build_registry(root)):
            errors.append(f"{docs_dir()}/audit/handoff.md not aligned with its HANDOFF_*.md "
                          "sources: run 'sdlc_check.py index'. A merge resolved by hand is "
                          "exactly what this catches")
        if len(workstreams) > REGISTRY_CAP:
            warnings.append(f"{len(workstreams)} open workstreams: the registry is meant to "
                            f"stay under {REGISTRY_CAP}")
        # Two files claiming one workstream is the collision this design does NOT
        # fix (two people opening the same work under different file names). It
        # would otherwise show up as two identical-looking rows and nothing else.
        seen = {}
        for p, meta in workstreams:
            wid = str(meta.get("workstream")).strip()
            if wid in seen:
                warnings.append(f"{docs_dir()}/audit/{p.name} and {seen[wid]} both claim "
                                f"workstream '{wid}': the registry shows two rows for one "
                                "workstream — decide which file owns it")
            else:
                seen[wid] = p.name
    if hand.is_file():
        m = re.search(r"(?:Date|Data):\s*(\d{4}-\d{2}-\d{2})", read_text(hand))
        if not m:
            warnings.append("audit/handoff.md without a 'Date: YYYY-MM-DD' header")
        else:
            try:
                stamp = datetime.strptime(m.group(1), "%Y-%m-%d").replace(tzinfo=timezone.utc)
                age = (datetime.now(timezone.utc) - stamp).days
                if age > 14:
                    warnings.append(f"audit/handoff.md is {age} days old: treat it as history, not current state")
            except ValueError:
                warnings.append("audit/handoff.md: date not parseable")

    for a in advisories:
        print(f"[note]  {a}")
    for w in warnings:
        print(f"[warn]  {w}")
    for e in errors:
        print(f"[ERROR] {e}")
    print(f"\nValidation: {len(errors)} errors, {len(warnings)} warnings, "
          f"{len(advisories)} advisories.")
    if advisories:
        print("[note] advisories are freshness signals: never fail a build, not even --strict.")
    if strict and warnings and not errors:
        print("[strict] warnings are failures in --strict mode.")
    return 1 if errors or (strict and warnings) else 0


# ------------------------------------------------------------- audit_plan

def parse_audit_plan(root):
    f = ai_path(root, "audit", "audit_plan.md")
    rows, lines = [], []
    if f.is_file():
        lines = read_text(f).splitlines()
        for i, line in enumerate(lines):
            if not line.strip().startswith("|"):
                continue
            cells = [c.strip() for c in line.strip().strip("|").split("|")]
            if len(cells) < 2:
                continue
            if cells[0].lower() in ("path", "percorso") or set(cells[0]) <= set("-: "):
                continue
            rows.append({
                "line": i,
                "path": cells[0],
                "status": cells[1].upper(),
                "ref": cells[2] if len(cells) > 2 else "",
                "note": cells[3] if len(cells) > 3 else "",
            })
    return f, lines, rows


def cmd_stale(root, hybrid=False):
    rc = 0
    # --- guide freshness (source_hash vs snapshot) — runs in EVERY mode
    drifted = []
    for rel, p, meta, _ in list_guides(root):
        df, rec = meta.get("distilled_from", ""), meta.get("source_hash", "")
        if not df or not rec:
            continue  # structure problems are validate's job
        src = root / df
        if not src.is_file():
            print(f"[warn]  {rel}: distilled_from '{df}' not found — snapshot missing")
            rc = 1
            continue
        if sha256_file(src) != rec:
            drifted.append((rel, df))
    for rel, df in drifted:
        print(f"[stale] {rel}: source snapshot '{df}' changed since distillation — regenerate the guide")
    if drifted:
        rc = 1
    # --- audit-plan staleness — delegated to devPNT/KL in hybrid
    if hybrid:
        print("[info] hybrid mode: audit-plan staleness is delegated to devPNT/KL, skipping.")
        return rc                                  # was: implicit skip-all; guide rc survives
    f, _, rows = parse_audit_plan(root)
    if not rows:
        print(f"[info] no rows in {f}: nothing to check "
              "(audit not initialized, or Hybrid mode where mapping is delegated to devPNT).")
        return rc                                  # was: return 0 — MUST carry guide rc
    use_git = git_available(root)
    stale = []
    for row in rows:
        if row["status"] != "ANALYZED":
            continue
        rel, ref = row["path"], row["ref"]
        # Confine BEFORE touching the filesystem: audit_plan.md is document
        # content, so an absolute row ('/'), a drive-relative one or a '..'
        # escape would otherwise walk outside the project (P-TM T2/T3). A bare
        # '/' is the row init.js used to seed, and `root / "/"` is the drive.
        target = confine_under(root, rel)
        if target is None:
            print(f"[warn]  {rel}: path is absolute, contains '..', or resolves outside "
                  "the project root: rejected (use a project-relative path, '.' for the root)")
            continue
        if not target.exists():
            print(f"[warn]  {rel}: path does not exist")
            continue
        changed = []
        if use_git and re.fullmatch(r"[0-9a-fA-F]{7,40}", ref or ""):
            res = git_changed_since(root, ref, rel.replace("\\", "/"))
            if res is None:
                print(f"[warn]  {rel}: git ref '{ref}' unresolvable, cannot evaluate")
                continue
            changed = res
        else:
            ts = parse_iso(ref)
            if ts is None:
                print(f"[warn]  {rel}: reference '{ref}' not parseable (neither git hash nor ISO UTC)")
                continue
            for fp in iter_files(target):
                mtime = datetime.fromtimestamp(fp.stat().st_mtime, tz=timezone.utc)
                if mtime > ts + MTIME_GRACE:
                    try:
                        name = str(fp.relative_to(root)).replace("\\", "/")
                    except ValueError:   # symlink out of the tree: report absolute, never crash
                        name = str(fp)
                    changed.append(name)
        if changed:
            stale.append((rel, changed))

    if not stale:
        print("[ok] no analyzed area was modified after its last recorded analysis.")
        return rc                                  # was: return 0 — MUST carry guide rc
    print("Areas modified after the last recorded analysis:")
    for rel, changed in stale:
        print(f"  {rel}  ({len(changed)} files)")
        for c in changed[:10]:
            print(f"    - {c}")
        if len(changed) > 10:
            print(f"    ... and {len(changed) - 10} more")
    print("\nAfter re-analyzing, record it with: sdlc_check.py mark <path>")
    return 1                                       # stale areas dominate: rc already implied


def cmd_mark(root, paths):
    if not require_ai_docs(root, "mark"):
        return 1
    f, lines, rows = parse_audit_plan(root)
    use_git_ref = git_available(root) and not any(
        git_has_changes(root, raw.replace("\\", "/").rstrip("/")) for raw in paths
    )
    ref = git_head(root) if use_git_ref else utc_now_iso()
    by_path = {r["path"].replace("\\", "/").rstrip("/"): r for r in rows}

    if not lines:
        lines = ["# Audit Plan", "",
                 "| Path | Status | Reference | Notes |",
                 "|---|---|---|---|"]
        rows = []

    def row_text(path, note):
        return f"| {path} | ANALYZED | {ref} | {note} |"

    # validate EVERY path before printing or mutating anything: a rejection
    # after an '[ok] ... added as ANALYZED' line is a lie the agent will act on
    keys = []
    for raw in paths:
        key = raw.replace("\\", "/").rstrip("/") or "."
        if confine_under(root, key) is None:
            print(f"[ERROR] {raw}: absolute, '..'-escaping, or outside the project root: "
                  "refusing to mark (use a project-relative path, '.' for the root). "
                  "Nothing was written.")
            return 1
        keys.append(key)

    appended = []
    for key in keys:
        display = key + ("/" if (root / key).is_dir() and key != "." else "")
        existing = by_path.get(key)
        if existing:
            lines[existing["line"]] = row_text(existing["path"], existing["note"])
            print(f"[ok] {existing['path']} -> ANALYZED ({ref})")
        else:
            appended.append(row_text(display, ""))
            print(f"[ok] {display} added as ANALYZED ({ref})")

    if appended:
        insert_at = (max(r["line"] for r in rows) + 1) if rows else len(lines)
        lines[insert_at:insert_at] = appended

    f.parent.mkdir(parents=True, exist_ok=True)
    f.write_text("\n".join(lines) + "\n", encoding="utf-8")
    return 0


def cmd_check(root, strict=False, hybrid=False):
    print("===== validate =====")
    rc_v = cmd_validate(root, strict=strict, hybrid=hybrid)
    print("\n===== stale =====")
    rc_s = cmd_stale(root, hybrid=hybrid)
    print(f"\ncheck: {'CLEAN' if not (rc_v or rc_s) else 'NOT CLEAN'} "
          f"(validate rc={rc_v}, stale rc={rc_s})")
    return 1 if (rc_v or rc_s) else 0


# --------------------------------------------------------------------- gate

def cmd_gate(args):
    file_path = args.file or ""
    if args.hook:
        try:
            # bytes -> utf-8-sig: the hook payload is UTF-8 JSON regardless of the
            # console code page; '-sig' strips the BOM (PowerShell pipes)
            raw = sys.stdin.buffer.read().decode("utf-8-sig", errors="replace")
            payload = json.loads(raw)
            file_path = (payload.get("tool_input") or {}).get("file_path") or ""
        except Exception:
            return 0  # unparseable input: do not block
    if not file_path:
        return 0
    root = Path(args.root).resolve() if args.root else find_project_root()
    try:
        rel = str(Path(file_path).resolve().relative_to(root)).replace("\\", "/")
    except ValueError:
        return 0  # outside the project: not this gate's concern
    if rel.startswith((docs_dir() + "/", "tests/", "test/")):
        return 0
    protected = [p.strip().replace("\\", "/").rstrip("/")
                 for p in (args.protected or "").split(";") if p.strip()]
    if not protected:
        return 0
    if not any(rel == p or rel.startswith(p + "/") for p in protected):
        return 0
    for _, meta, _ in list_analyses(root):
        if meta.get("status") == "IN_PROGRESS":
            return 0
    if args.hybrid and has_etdd_shadow(root):
        return 0  # Hybrid design gate: an approved E-TDD shadow authorizes the change
    if args.hybrid:
        sys.stderr.write(
            f"[sdlc gate] '{rel}' is on a protected path but no E-TDD shadow "
            "(solutions/SHADOW_*tdd*.md) exists and no ANALYSIS_*.md is IN_PROGRESS. "
            "In Hybrid mode, export the approved E-TDD shadow from devPNT before implementing.\n")
        return 2
    sys.stderr.write(
        f"[sdlc gate] '{rel}' is on a protected path but no ANALYSIS_*.md is IN_PROGRESS. "
        "If your analysis already exists, set its frontmatter to 'status: IN_PROGRESS' "
        "(that flip is what opens the gate); otherwise write it first (Phase 3).\n")
    return 2


# --------------------------------------------------------------------- plan
# Subagent Execution (Feature A). Zero-execution surface: this section and
# everything it calls MUST NOT spawn a process (no subprocess/os.system/eval/
# exec, no git_* helper). It validates a PLAN_[feature].md and prints a task
# brief as text; the orchestrator (dispatch.md) is the sole executor.

_PLAN_JSON_RE = re.compile(r"```json\s*\n(.*?)```", re.DOTALL)


def extract_plan_json(text):
    """Extract the first fenced ```json block from a PLAN_[feature].md body.
    Returns (data, "") on success, or (None, reason) on any failure. Never
    raises: a malformed or missing block is a validation failure, not a crash."""
    m = _PLAN_JSON_RE.search(text or "")
    if not m:
        return None, "no fenced ```json block found in the plan file"
    try:
        data = json.loads(m.group(1))
    except (ValueError, TypeError) as e:
        return None, f"malformed JSON in the plan block: {e}"
    if not isinstance(data, dict):
        return None, "plan JSON block must be a JSON object"
    return data, ""


def load_ledger(path):
    """Read the sidecar ledger {"<task_id>": {"status", "verify_result",
    "timestamp"}}. Absent file -> ({}, ""). Malformed/unreadable -> ({}, reason).
    Never raises, never hangs: the ledger is untrusted state read on every call."""
    if not path.is_file():
        return {}, ""
    try:
        raw = read_text(path)
        data = json.loads(raw)
    except (ValueError, TypeError, OSError) as e:
        return {}, f"ledger '{path}' unreadable/malformed, treating as empty: {e}"
    if not isinstance(data, dict):
        return {}, f"ledger '{path}' is not a JSON object, treating as empty"
    return data, ""


def _confine_or_reject(base, rel, label, rel_label, errors):
    t = confine_under(base, rel)
    if t is None:
        errors.append(f"{rel_label}: {label} '{rel}' is absolute, contains '..', or escapes "
                      f"'{base}' — rejected (fail closed)")
    return t


def _validate_plan_tasks(root, data, rel_label, errors, warnings):
    """Shared core of `plan validate`/`plan brief`: schema + confinement checks.
    Returns the task list (possibly empty) on success; errors/warnings are
    appended in place. Callers decide the exit code."""
    tasks = data.get("tasks")
    if not isinstance(tasks, list) or not tasks:
        errors.append(f"{rel_label}: 'tasks' must be a non-empty JSON array")
        return []
    ref_dir = ai_path(root, "reference")
    kb_ref = DEFAULT_KB_ROOT / "ai_docs" / "reference"
    seen_ids = set()
    for i, task in enumerate(tasks):
        loc = f"{rel_label}: task[{i}]"
        if not isinstance(task, dict):
            errors.append(f"{loc}: not a JSON object")
            continue
        missing = [k for k in PLAN_TASK_REQUIRED if not task.get(k)]
        if missing:
            errors.append(f"{loc}: missing required field(s): {', '.join(missing)}")
        if not task.get("paths") and not task.get("produces"):
            errors.append(f"{loc}: must declare at least one of 'paths'/'produces'")
        tid = task.get("id")
        if tid:
            if tid in seen_ids:
                errors.append(f"{loc}: duplicate task id '{tid}'")
            seen_ids.add(tid)
        for key in ("paths", "consumes", "produces"):
            for p in (task.get(key) or []):
                _confine_or_reject(root, p, key, loc, errors)
        for g in (task.get("guides") or []):
            in_project = confine_under(ref_dir, g)
            in_kb = confine_under(kb_ref, g)
            if in_project is None and in_kb is None:
                errors.append(f"{loc}: guide '{g}' is not confined under the project reference "
                              f"dir ({ref_dir}) or the agent KB reference dir ({kb_ref}) — rejected")
    return tasks


def cmd_plan(root, args):
    """Zero-execution: validates/briefs a PLAN_[feature].md. Never spawns a
    process, never calls a git_* helper, never runs the opaque `verify` text —
    it is printed, not executed."""
    plan_path = Path(args.file)
    if not plan_path.is_absolute():
        plan_path = root / plan_path
    if not plan_path.is_file():
        sys.stderr.write(f"[plan] plan file not found: {plan_path}\n")
        return 2
    rel_label = str(plan_path)
    data, reason = extract_plan_json(read_text(plan_path))
    if data is None:
        sys.stderr.write(f"[plan] {rel_label}: {reason}\n")
        return 2

    errors, warnings = [], []
    tasks = _validate_plan_tasks(root, data, rel_label, errors, warnings)

    ledger_path = plan_path.with_name(plan_path.stem + ".ledger.json")
    ledger, ledger_reason = load_ledger(ledger_path)
    if ledger_reason:
        warnings.append(ledger_reason)
    if not errors:
        task_ids = {t.get("id") for t in tasks if isinstance(t, dict)}
        for lid in ledger:
            if lid not in task_ids:
                warnings.append(f"ledger id '{lid}' not found in {rel_label}: orphaned entry (not fatal)")

    for w in warnings:
        sys.stderr.write(f"[warn] {w}\n")
    for e in errors:
        sys.stderr.write(f"[ERROR] {e}\n")

    if args.plan_cmd == "validate":
        if errors:
            sys.stderr.write(f"\n[plan] validate: {len(errors)} errors, {len(warnings)} warnings.\n")
            return 2
        print(f"[ok] {rel_label}: plan valid ({len(tasks)} task(s), {len(warnings)} warning(s)).")
        return 0

    # brief
    if errors:
        sys.stderr.write(f"\n[plan] brief: plan is invalid, refusing to brief ({len(errors)} errors).\n")
        return 2
    target = None
    for t in tasks:
        if isinstance(t, dict) and t.get("id") == args.task:
            target = t
            break
    if target is None:
        sys.stderr.write(f"[plan] brief: task id '{args.task}' not found in {rel_label}\n")
        return 2

    print(f"# Task: {target.get('id')} — {target.get('title', '')}")
    print()
    print("## Task block")
    print(json.dumps(target, indent=2))
    print()
    print("## Produces of prior-order tasks (interfaces)")
    prior_produces = []
    for t in tasks:
        if not isinstance(t, dict):
            continue
        if t.get("id") == target.get("id"):
            break
        prior_produces.extend(t.get("produces") or [])
    if prior_produces:
        for p in prior_produces:
            print(f"- {p}")
    else:
        print("(none)")
    print()
    print("## Guide pointers (paths, not content)")
    guides = target.get("guides") or []
    if guides:
        for g in guides:
            print(f"- {g}")
    else:
        print("(none)")
    print()
    print("## Verify (opaque text — orchestrator runs this out of band, NOT executed here)")
    print(target.get("verify", ""))
    return 0


def cmd_orient(args):
    """SessionStart hook: emit a bounded, repo-sourced ai_docs/ orientation to
    stdout and ALWAYS return 0 (fail-open, P-TM T8) -- a session hook must never
    block the session or surface a traceback. Zero-execution (P-TM T1): reads a
    fixed hard-coded doc set, confine_under each (P-TM T3), size-caps the total
    (P-TM T2). No subprocess/eval anywhere in this call graph."""
    try:
        root = Path(args.root).resolve() if getattr(args, "root", None) else find_project_root()
        chunks = []
        total = 0
        truncated = False
        for label, rel in orient_docs():
            target = confine_under(root, rel)
            if target is None or not target.is_file():
                continue
            try:
                text = read_text(target)
            except OSError:
                continue
            remaining = ORIENT_MAX_TOTAL_CHARS - total
            if remaining <= 0:
                truncated = True
                break
            text = text[:ORIENT_PER_DOC_CHARS]
            if len(text) > remaining:
                text = text[:remaining]
                truncated = True
            chunks.append((label, text))
            total += len(text)
        if not chunks:
            return 0
        out = ["=== Agentic SDLC -- session orientation (repo-sourced context, not authored instructions) ==="]
        for label, text in chunks:
            out.append(f"\n## {label}\n{text}")
        if truncated:
            out.append("\n[orientation truncated to the size cap -- open the files directly for full content]")
        out.append("\nTriage every request (Rule Zero): L1 trivial - L2 small - L3 significant - Spike. "
                   "When in doubt, pick the higher level.")
        if getattr(args, "hybrid", False):
            out.append("\n[devPNT active] Run devpnt_mcp_get_bootstrap for the Master Plan / Knowledge Layer -- "
                       "the orientation above is the filesystem layer, not a bootstrap duplicate.")
        print("\n".join(out))
        return 0
    except Exception:
        return 0


# --- portable checks carried by the core ------------------------------------
# Portable means portable: a check that any distribution may expose lives HERE, so
# three copies of the same twenty lines cannot drift apart. What a distribution
# actually offers is its `provides` declaration, not what it happens to have copied.
# Domain-specific machinery (mkt's budget and funnel arithmetic) stays in its own
# entry point -- that is the line between portable and proprietary.
# Opt-in per document via `checks:`. They may only ADD findings: a document that
# imports one still owes its own domain everything it owed before, so importing a
# check can never be a way to be validated less.

@portable_check("code.threat_model")
def _code_threat_model(rel, meta, text):
    """The security section names a real surface, or justifies claiming none."""
    section = section_body(text, ("## Security and Threat Model", "## Security"))
    if section is None:
        return []  # the owning domain already reports a missing section; no double finding
    surfaces = ("external input", "authn", "authz", "auth", "crypto", "network",
                "personal data", "filesystem", "supply chain")
    low = section.lower()
    if any(s in low for s in surfaces):
        return []
    if "no security impact" in low or "no new security" in low:
        if len(section.split()) < 15:
            return [("warning", "'no security impact' is declared, not justified: "
                                "say which surfaces you checked and why none is touched")]
        return []
    return [("warning", "no security surface named (external input, authN/authZ, crypto, "
                        "network, personal data, filesystem, supply chain) and no justified "
                        "claim that none is touched")]


@portable_check("knowledge.sources")
def _knowledge_sources(rel, meta, text):
    """A knowledge artifact says what it was written from AND how that was verified."""
    section = section_body(text, ("## Sources and Verification",))
    if section is None:
        return []
    findings = []
    if not re.search(r"(?m)^\s*[-*|]|\bhttps?://|\.(?:md|pdf|docx?|csv|xlsx?)\b", section):
        findings.append(("warning", "no source is named: a distillation whose origin cannot "
                                    "be reopened is model knowledge, not knowledge work"))
    if not re.search(r"verif|cross-check|checked against|confirmed", section, re.I):
        findings.append(("warning", "sources are listed but not verified: say how each was "
                                    "confirmed, or mark explicitly what could not be"))
    return findings


# ------------------------------------------------------------------- migrate

def _migration_plan(root, src_name, dst_name):
    """(files, refs, conflicts) for a docs-root move. Reads only."""
    src = root / src_name
    dst = root / dst_name
    files, refs, conflicts = [], [], []
    if not src.is_dir():
        return None, None, None
    for p in sorted(src.rglob("*")):
        if not p.is_file():
            continue
        rel = p.relative_to(src)
        target = dst / rel
        if target.exists():
            conflicts.append(rel.as_posix())
        files.append(rel.as_posix())
        if p.suffix.lower() in (".md", ".txt", ".json", ".yml", ".yaml"):
            try:
                if f"{src_name}/" in p.read_text(encoding="utf-8", errors="replace"):
                    refs.append(rel.as_posix())
            except OSError:
                pass
    return files, refs, conflicts


def _external_references(root, src_name):
    """Files OUTSIDE both roots that mention the old root.

    Reported, never edited: the protocol pointers (CLAUDE.md, AGENTS.md,
    .cursorrules) are user-authored, and a tool that rewrites them is the one thing
    `init` has refused to do since day one."""
    out = []
    for p in sorted(root.rglob("*")):
        if not p.is_file() or p.suffix.lower() not in (".md", ".txt", ".json", ".yml", ".yaml"):
            continue
        rel = p.relative_to(root)
        if rel.parts and rel.parts[0] in (src_name, ".git"):
            continue
        if any(part in SKIP_DIRS for part in rel.parts):
            continue
        try:
            if f"{src_name}/" in p.read_text(encoding="utf-8", errors="replace"):
                out.append(rel.as_posix())
        except OSError:
            pass
    return out


def cmd_migrate(root, args):
    """Relocate the documentation root. Dry-run by default; never deletes.

    Reversibility is structural, not promised: the old root is COPIED, never moved,
    so undoing the migration is deleting the new directory. Both roots stay
    readable throughout, which is what lets a team switch over one session at a
    time instead of in one jump.
    """
    src_name = args.from_dir
    dst_name = args.to_dir
    if src_name == dst_name:
        print(f"[ERROR] --from and --to are both '{src_name}': nothing to do.")
        return 1
    src = root / src_name
    if not src.is_dir():
        print(f"[ERROR] {src} not found: nothing to migrate.")
        return 1

    files, refs, conflicts = _migration_plan(root, src_name, dst_name)
    external = _external_references(root, src_name)

    print(f"=== migrate {src_name}/ -> {dst_name}/ ===")
    print(f"  {len(files)} file(s) to copy, {len(refs)} carrying '{src_name}/' references")
    for rel in files[:20]:
        print(f"    {src_name}/{rel}  ->  {dst_name}/{rel}")
    if len(files) > 20:
        print(f"    ... and {len(files) - 20} more")
    if conflicts:
        print(f"\n[ERROR] {len(conflicts)} file(s) already exist under {dst_name}/:")
        for rel in conflicts[:10]:
            print(f"    {dst_name}/{rel}")
        print("  Refusing to overwrite. Move them aside, or migrate into an empty root.")
        return 1
    if external:
        print(f"\n[note] {len(external)} file(s) OUTSIDE the docs roots mention "
              f"'{src_name}/'. They are NOT touched -- protocol pointers and READMEs are "
              "yours to edit:")
        for rel in external[:10]:
            print(f"    {rel}")

    if not args.apply:
        print("\n[dry-run] Nothing was written. Re-run with --apply to perform the copy.")
        print("          The old root is kept: this migration is undone by deleting "
              f"{dst_name}/.")
        return 0

    if git_available(root) and git_has_changes(root, "."):
        print("\n[ERROR] the working tree has uncommitted changes. Commit or stash first: "
              "a migration you cannot diff is a migration you cannot review.")
        return 1

    written = 0
    for rel in files:
        source = src / rel
        target = root / dst_name / rel
        target.parent.mkdir(parents=True, exist_ok=True)
        if rel in refs:
            text = source.read_text(encoding="utf-8", errors="replace")
            target.write_text(text.replace(f"{src_name}/", f"{dst_name}/"), encoding="utf-8")
        else:
            target.write_bytes(source.read_bytes())
        written += 1
    print(f"\n[ok] copied {written} file(s) into {dst_name}/. "
          f"{src_name}/ is untouched -- delete it yourself once you are satisfied.")
    print(f"     Next: run `validate --docs-dir {dst_name}` and compare it with "
          f"`validate --docs-dir {src_name}`. They should say the same thing.")
    return 0



# --------------------------------------------------------------------- main

def main(argv=None):
    common = argparse.ArgumentParser(add_help=False)
    common.add_argument("--root", help="project root (default: walk up until a docs root is found)")
    common.add_argument("--docs-dir", dest="docs_dir",
                        help="name of the documentation root (default: ai_docs; "
                             "use it to reach a legacy root such as mkt_docs, "
                             "or to disambiguate a half-migrated tree)")

    strict_opt = argparse.ArgumentParser(add_help=False)
    strict_opt.add_argument("--strict", action="store_true",
                            help="fail on warnings and on a missing docs root (for CI)")

    hybrid_opt = argparse.ArgumentParser(add_help=False)
    hybrid_opt.add_argument("--hybrid", action="store_true",
                            help="Hybrid/devPNT mode: audit-plan staleness is delegated to devPNT/KL; "
                                 "the gate also unlocks on an E-TDD shadow")

    ap = argparse.ArgumentParser(prog="sdlc_check.py",
                                 description="Mechanical validator for Agentic SDLC")
    sub = ap.add_subparsers(dest="cmd", required=True)
    sub.add_parser("check", parents=[common, strict_opt, hybrid_opt],
                   help="closure gate: validate + stale in one command")
    sub.add_parser("validate", parents=[common, strict_opt, hybrid_opt],
                   help="verify docs-root coherence")
    sub.add_parser("index", parents=[common], help="regenerate features_history.md + ai_docs/INDEX.md")
    sub.add_parser("stale", parents=[common, hybrid_opt], help="areas modified after the last analysis")
    gp_mig = sub.add_parser("migrate", parents=[common],
                            help="relocate the documentation root (dry-run by default)")
    gp_mig.add_argument("--from", dest="from_dir", required=True,
                        help="current docs root name, e.g. mkt_docs")
    gp_mig.add_argument("--to", dest="to_dir", default=DEFAULT_DOCS_DIR,
                        help=f"target docs root name (default: {DEFAULT_DOCS_DIR})")
    gp_mig.add_argument("--apply", action="store_true",
                        help="actually copy (default is a dry run; never deletes)")

    mp = sub.add_parser("mark", parents=[common], help="record paths as ANALYZED")
    mp.add_argument("paths", nargs="+", help="paths relative to the project root")
    gp = sub.add_parser("gate", parents=[common, hybrid_opt], help="PreToolUse hook (exit 2 = block)")
    gp.add_argument("--hook", action="store_true", help="read the hook JSON payload from stdin")
    gp.add_argument("--file", help="file path to evaluate (alternative to --hook)")
    gp.add_argument("--protected", default="", help="protected prefixes separated by ';' (e.g. \"src/auth;src/crypto\")")

    sub.add_parser("orient", parents=[common, hybrid_opt],
                   help="SessionStart hook: emit docs-root orientation to stdout (fail-open, zero-execution)")

    pp = sub.add_parser("plan", parents=[common],
                        help="Subagent Execution: validate/brief a PLAN_[feature].md (zero-execution)")
    pp_sub = pp.add_subparsers(dest="plan_cmd", required=True)
    pv = pp_sub.add_parser("validate", help="schema + confinement + ledger cross-check (exit 2 on error)")
    pv.add_argument("file", help="path to the PLAN_[feature].md file")
    pb = pp_sub.add_parser("brief", help="print a task's brief to stdout (verify text is NOT executed)")
    pb.add_argument("file", help="path to the PLAN_[feature].md file")
    pb.add_argument("--task", required=True, help="task id to brief")

    args = ap.parse_args(argv)

    # Resolve the documentation root ONCE, before any command runs, so every
    # surface -- paths, messages, generated headers, the gate's exempt prefix --
    # names the same thing.
    try:
        discovered, name = resolve_docs_dir(args, getattr(args, "root", None))
    except AmbiguousDocsRoot as exc:
        if args.cmd == "migrate":
            # A tree with two roots is not an obstacle to `migrate`: it is the state
            # `migrate` exists to resolve, and both names come from --from/--to, so
            # nothing is being guessed. Refusing here would make the guard block the
            # one command that ends the ambiguity.
            discovered, name = None, args.from_dir
        elif args.cmd == "orient":
            # The SessionStart hook is fail-open by contract: it must never block a
            # session, not even on a half-migrated tree. It orients on the default
            # and says so rather than exiting non-zero.
            print(f"[note] {exc}")
            discovered, name = None, DEFAULT_DOCS_DIR
        else:
            print(f"[ERROR] {exc}")
            return 1
    set_docs_dir(name)

    if args.cmd == "gate":
        return cmd_gate(args)
    if args.cmd == "orient":
        return cmd_orient(args)

    root = Path(args.root).resolve() if args.root else (discovered or find_project_root())
    if args.cmd == "check":
        return cmd_check(root, strict=args.strict, hybrid=args.hybrid)
    if args.cmd == "validate":
        return cmd_validate(root, strict=args.strict, hybrid=args.hybrid)
    if args.cmd == "index":
        return cmd_index(root)
    if args.cmd == "stale":
        return cmd_stale(root, hybrid=args.hybrid)
    if args.cmd == "mark":
        return cmd_mark(root, args.paths)
    if args.cmd == "migrate":
        return cmd_migrate(root, args)
    if args.cmd == "plan":
        return cmd_plan(root, args)
    return 0


if __name__ == "__main__":
    sys.exit(main())
