#!/usr/bin/env python3
"""
uca.py — reference parser and validator for Universal Cognitive Architecture (UCA 1.0) addresses.

One file, no dependencies, Python 3.10+.

    python uca.py MA50_01_SY20            # parse and explain one or more addresses
    python uca.py --check corpus.txt      # validate a file of addresses, one per line
    python uca.py --json IN50_2026_06_A01 # machine-readable output
    python uca.py init ./library          # write the First Fifteen as skeleton files

The grammar in the specification is written in ABNF-style and is ambiguous on its own
(a two-digit segment can be a structural part, a month, or a day). The specification
resolves that with two reading rules that live in prose. This parser encodes those rules
positionally, so every address has exactly one parse or a stated reason it has none.

Positional rules, as implemented (each is cited to the section of the spec it comes from):

  base        XX##, where XX is one of the eleven Organizational cognates and ## is 00–99,
              or the reserved prefix AI with ## 00–19 (the reader's instructions; AI00 is
              the kernel). AI takes parts and EATS, never a tracker.      (Valid Prefixes)
              Tier by number: 00–19 System, 20–49 Canon, 50–79 Projects, 80–99 Patterns.
              Canon is partitioned: 20–34 leaves, 35–49 masters.        (Range Map)
  part        two digits, _00 to _99, a structural part, unless its parent is a year (then a
              month) or a month (then a day). 00 is the foundation slot: the one document or
              folder that governs the series beneath it.                   (Reading rule 1)
  EATS        E##, A##, T##; S## is always a sprint — beneath a year, that year's Nth
              sprint; elsewhere, the Nth sprint of its parent. 00 is the series' foundation.
              A record of what occurred files beneath its date by E/A/T,
              never by ordinal.                                            (Reading rule 2)
  temporal    a four-digit year; beneath a year: Q1–Q4, a month, W01–W53, or a sprint;
              beneath a month: a day. Every grain hangs from the year.    (Temporal Extensions)
  tracker     three uppercase letters directly beneath the base or a part, and never
              beneath canon. It may stand alone as a root, or take a year.  (Temporal Extensions)
  foreign     a whole address hosted beneath this one. It begins at the first segment
              shaped like a base, it is always the final segment, and it is permitted
              only beneath active work (50–79).                            (Hosted Addresses)
  leaf        a canon leaf (20–34) takes no structural parts; it is promoted to a master
              rather than given children.                                 (Range Map)
"""

from __future__ import annotations

import argparse
import json
import re
import sys
from dataclasses import dataclass, field, asdict
from enum import Enum
from pathlib import Path
from typing import Iterable

__version__ = "1.0.1"

# ----------------------------------------------------------------------------- vocabulary

ORGANIZATIONAL: dict[str, str] = {
    "ID": "Identity",
    "MA": "Market",
    "IN": "Innovation",
    "BR": "Brand",
    "PD": "Productivity",
    "WF": "Workflow",
    "PR": "Process",
    "CU": "Culture",
    "SY": "System",
    "LA": "Language",
    "IT": "Integration",
}

# Reserved in UCA 1.0: admitted, not a cognate. OS range only; addresses the reader.
RESERVED: dict[str, str] = {
    "AI": "AI (reserved: operating instructions for the reader)",
}

# Illustrative in UCA 1.0: shown in the specification, not yet standardized.
ILLUSTRATIVE: dict[str, str] = {
    "LP": "Life Profile (Personal)",
    "PF": "Personal Finance (Personal)",
    "CS": "Customer (Marketplace)",
    "CP": "Competitor (Marketplace)",
    "EC": "Ecosystem (Marketplace)",
    "SI": "Signal (Marketplace)",
}

FIRST_FIFTEEN: dict[str, str] = {
    "SY20": "Context Node",
    "ID20": "Cultural Values",
    "ID21": "Product Values",
    "ID22": "Customer Values",
    "ID23": "Market Values",
    "MA20": "Competitive Position",
    "MA21": "Business Model",
    "MA22": "Revenue Strategy",
    "IN20": "Product Roadmap",
    "IN21": "Growth Strategy",
    "BR20": "Brand Positioning",
    "BR21": "Market Expression",
    "BR22": "Stakeholder Communications",
    "LA20": "Empirical Metrics",
    "LA21": "Empathic Metrics",
}

MINIMUM_VIABLE_INTELLIGENCE = ("SY20", "ID20", "ID21", "LA20", "LA21")


class Tier(str, Enum):
    SYSTEM = "System"      # 00–19  OS
    CANON = "Canon"        # 20–49  PMNs
    PROJECTS = "Projects"  # 50–79  AMNs
    PATTERNS = "Patterns"  # 80–99  LMNs

    @classmethod
    def of(cls, number: int) -> "Tier":
        if number < 20:
            return cls.SYSTEM
        if number < 50:
            return cls.CANON
        if number < 80:
            return cls.PROJECTS
        return cls.PATTERNS


class Kind(str, Enum):
    BASE = "base"
    PART = "part"
    EXPLORATION = "exploration"
    ANSWER = "answer"
    TEMPLATE = "template"
    SPRINT = "sprint"
    YEAR = "year"
    QUARTER = "quarter"
    MONTH = "month"
    WEEK = "week"
    DAY = "day"
    TRACKER = "tracker"
    FOREIGN = "foreign"


EATS_KINDS = {"E": Kind.EXPLORATION, "A": Kind.ANSWER, "T": Kind.TEMPLATE, "S": Kind.SPRINT}
TEMPORAL_KINDS = {Kind.YEAR, Kind.QUARTER, Kind.MONTH, Kind.WEEK, Kind.DAY}

# ----------------------------------------------------------------------------- shapes

_BASE = re.compile(r"^([A-Z]{2})(\d{2})$")
_TWO = re.compile(r"^\d{2}$")
_EATS = re.compile(r"^([EATS])(\d{2})$")
_YEAR = re.compile(r"^(19|2\d)\d{2}$")
_QUARTER = re.compile(r"^Q([1-4])$")
_WEEK = re.compile(r"^W(\d{2})$")
_TRACKER = re.compile(r"^[A-Z]{3}$")
_ADDRESS_CHARS = re.compile(r"^[A-Z0-9_]+$")


# ----------------------------------------------------------------------------- results


class UCAError(ValueError):
    """An address that does not conform. The message says which rule it broke."""


@dataclass(frozen=True)
class Segment:
    kind: Kind
    text: str
    value: int | str | None = None

    def __str__(self) -> str:
        return f"{self.text}({self.kind.value})"


@dataclass
class Address:
    """A parsed UCA address. Construct with parse(); the fields are read-only in spirit."""

    text: str
    cognate: str
    cognate_name: str
    number: int
    tier: Tier
    segments: list[Segment] = field(default_factory=list)
    host: "Address | None" = None
    illustrative: bool = False

    # --- derived facts a validator or a filing check would ask for

    @property
    def base(self) -> str:
        return f"{self.cognate}{self.number:02d}"

    @property
    def is_leaf(self) -> bool:
        return self.tier is Tier.CANON and self.number < 35

    @property
    def is_master(self) -> bool:
        return self.tier is Tier.CANON and self.number >= 35

    @property
    def is_constitutional(self) -> bool:
        return self.base in FIRST_FIFTEEN and not self.segments

    @property
    def title(self) -> str | None:
        return FIRST_FIFTEEN.get(self.base)

    @property
    def depth(self) -> int:
        return len(self.segments)

    @property
    def is_foundation(self) -> bool:
        """True when the last segment is a 00 slot: the document that governs its series."""
        last = self.segments[-1] if self.segments else None
        return last is not None and last.value == 0 and last.kind in (Kind.PART, *EATS_KINDS.values())

    def explain(self) -> str:
        """One line a person can read: what each segment is."""
        parts = [f"{self.base} = {self.cognate_name} · {self.tier.value}"]
        if self.title:
            parts[0] += f" · {self.title} (First Fifteen)"
        elif self.tier is Tier.CANON:
            parts[0] += " · leaf" if self.is_leaf else " · master"
        for seg in self.segments:
            if seg.kind is Kind.FOREIGN and self.host is not None:
                parts.append(f"{seg.text} = hosted address → {self.host.explain()}")
            elif seg.value == 0 and seg.kind in (Kind.PART, *EATS_KINDS.values()):
                parts.append(f"{seg.text} = {seg.kind.value} · foundation slot")
            else:
                parts.append(f"{seg.text} = {seg.kind.value}")
        return " ; ".join(parts)

    def to_dict(self) -> dict:
        d = asdict(self)
        d["tier"] = self.tier.value
        d["segments"] = [{"kind": s.kind.value, "text": s.text, "value": s.value} for s in self.segments]
        d["host"] = self.host.to_dict() if self.host else None
        d["base"] = self.base
        d["title"] = self.title
        return d


# ----------------------------------------------------------------------------- parser


def parse(text: str, *, allow_illustrative: bool = False) -> Address:
    """
    Parse one address. Returns an Address or raises UCAError naming the rule broken.

    `allow_illustrative` admits the Personal and Marketplace prefixes the specification
    shows but does not standardize. A conformant implementation leaves it False.
    """
    raw = text.strip()
    if not raw:
        raise UCAError("empty address")
    if not _ADDRESS_CHARS.match(raw):
        raise UCAError(f"{raw!r}: addresses use only A–Z, 0–9 and underscores")

    tokens = raw.split("_")
    if any(t == "" for t in tokens):
        raise UCAError(f"{raw!r}: empty segment (doubled or trailing underscore)")

    base = _parse_base(tokens[0], allow_illustrative)
    address = Address(
        text=raw,
        cognate=base[0],
        cognate_name=base[1],
        number=base[2],
        tier=Tier.of(base[2]),
        illustrative=base[0] in ILLUSTRATIVE,
    )

    parent = Kind.BASE
    for i, tok in enumerate(tokens[1:], start=1):
        # A segment shaped like a base begins a hosted address; it consumes the rest.
        if _BASE.match(tok):
            if address.tier is not Tier.PROJECTS:
                raise UCAError(
                    f"{raw!r}: hosted address {tok!r} beneath {address.tier.value}; "
                    "hosting applies in active work (50–79) only, never canon"
                )
            if parent in TEMPORAL_KINDS or parent is Kind.TRACKER:
                raise UCAError(
                    f"{raw!r}: hosted address {tok!r} beneath a {parent.value}; "
                    "a foreign address hangs from the host path, not from a date"
                )
            foreign_text = "_".join(tokens[i:])
            host = parse(foreign_text, allow_illustrative=allow_illustrative)
            address.segments.append(Segment(Kind.FOREIGN, foreign_text, host.base))
            address.host = host
            return address

        seg = _parse_segment(tok, parent, address, raw)
        address.segments.append(seg)
        parent = seg.kind

    return address


def _parse_base(tok: str, allow_illustrative: bool) -> tuple[str, str, int]:
    m = _BASE.match(tok)
    if not m:
        raise UCAError(f"{tok!r}: a base is two uppercase letters and two digits (XX##)")
    code, number = m.group(1), int(m.group(2))
    if code in ORGANIZATIONAL:
        return code, ORGANIZATIONAL[code], number
    if code in RESERVED:
        if number > 19:
            raise UCAError(
                f"{tok!r}: the reserved prefix AI holds the OS range only (AI00–AI19); "
                "there is no AI canon, projects or patterns"
            )
        return code, RESERVED[code], number
    if code in ILLUSTRATIVE:
        if allow_illustrative:
            return code, ILLUSTRATIVE[code], number
        raise UCAError(
            f"{tok!r}: {code} is an illustrative prefix (Personal/Marketplace), not standardized "
            "in UCA 1.0; pass allow_illustrative=True to admit it"
        )
    raise UCAError(f"{tok!r}: {code} is not one of the eleven Organizational cognates or the reserved prefix AI")


def _parse_segment(tok: str, parent: Kind, address: Address, raw: str) -> Segment:
    """Classify one non-foreign segment by its shape and its parent."""

    # --- EATS: E##, A##, T## anywhere; S## is always a sprint.
    m = _EATS.match(tok)
    if m:
        kind = EATS_KINDS[m.group(1)]
        n = int(m.group(2))  # 00 is the series' foundation
        if kind is Kind.SPRINT and parent is Kind.TRACKER:
            raise UCAError(f"{raw!r}: a sprint cannot hang directly from a tracker; give it a year")
        return Segment(kind, tok, n)

    # --- tracker: three letters, at the root, never beneath canon.
    if _TRACKER.match(tok):
        if address.cognate in RESERVED:
            raise UCAError(f"{raw!r}: tracker {tok!r} beneath the reserved prefix AI; AI takes parts and EATS only")
        if address.tier is Tier.CANON:
            raise UCAError(f"{raw!r}: tracker {tok!r} beneath canon; canon (20–49) never takes one")
        if parent not in (Kind.BASE, Kind.PART):
            raise UCAError(
                f"{raw!r}: tracker {tok!r} beneath a {parent.value}; "
                "a tracker sits between the base (or a part) and a year"
            )
        return Segment(Kind.TRACKER, tok, tok)

    # --- year: four digits, at the root or beneath a tracker or a part.
    if _YEAR.match(tok):
        if parent in TEMPORAL_KINDS:
            raise UCAError(f"{raw!r}: year {tok!r} beneath a {parent.value}; a year is the spine, not a grain")
        return Segment(Kind.YEAR, tok, int(tok))

    # --- grains that only hang from a year.
    m = _QUARTER.match(tok)
    if m:
        _require_parent(parent, Kind.YEAR, tok, raw, "a quarter")
        return Segment(Kind.QUARTER, tok, int(m.group(1)))
    m = _WEEK.match(tok)
    if m:
        _require_parent(parent, Kind.YEAR, tok, raw, "a week")
        w = int(m.group(1))
        if not 1 <= w <= 53:
            raise UCAError(f"{raw!r}: week {tok!r} out of range W01–W53")
        return Segment(Kind.WEEK, tok, w)

    # --- two digits: part, month, or day, by reading rule 1.
    if _TWO.match(tok):
        n = int(tok)
        if parent is Kind.YEAR:
            if not 1 <= n <= 12:
                raise UCAError(f"{raw!r}: {tok!r} beneath a year is a month; months are 01–12")
            return Segment(Kind.MONTH, tok, n)
        if parent is Kind.MONTH:
            if not 1 <= n <= 31:
                raise UCAError(f"{raw!r}: {tok!r} beneath a month is a day; days are 01–31")
            return Segment(Kind.DAY, tok, n)
        if parent in (Kind.QUARTER, Kind.WEEK, Kind.DAY):
            raise UCAError(f"{raw!r}: {tok!r} beneath a {parent.value}; nothing numeric hangs from that grain")
        if parent is Kind.TRACKER:
            raise UCAError(f"{raw!r}: {tok!r} beneath a tracker; a tracker takes a year, not a part")
        # n == 0 is the foundation slot: the document or folder that governs the series.
        if address.is_leaf and parent is Kind.BASE:
            raise UCAError(
                f"{raw!r}: structural part beneath canon leaf {address.base}; "
                "a leaf (20–34) is promoted to a master (35–49), never given children"
            )
        return Segment(Kind.PART, tok, n)

    raise UCAError(f"{raw!r}: segment {tok!r} matches no pattern in the grammar")


def _require_parent(parent: Kind, wanted: Kind, tok: str, raw: str, what: str) -> None:
    if parent is not wanted:
        raise UCAError(f"{raw!r}: {tok!r} — {what} hangs from a year, not from a {parent.value}")


# ----------------------------------------------------------------------------- convenience


def is_valid(text: str, **kw) -> bool:
    try:
        parse(text, **kw)
        return True
    except UCAError:
        return False


def check(lines: Iterable[str], **kw) -> tuple[list[Address], list[tuple[str, str]]]:
    """Validate many addresses. Returns (parsed, [(text, reason), ...])."""
    good: list[Address] = []
    bad: list[tuple[str, str]] = []
    for line in lines:
        s = line.split("#", 1)[0].strip()
        if not s:
            continue
        try:
            good.append(parse(s, **kw))
        except UCAError as e:
            bad.append((s, str(e)))
    return good, bad


# ----------------------------------------------------------------------------- init

_SKELETON = """\
# {address} {title}

**Address:** {address}
**Tier:** Canon (Persistent Memory)
**Version:** 1.0
**Status:** Draft
**Owner:**
**Approved by:**
**Effective:**

## What this memory is

{about}

## Content

<!-- Write it here, in plain language. This file is the memory; the address is its meaning. -->
"""

_ABOUT = {
    "SY20": "Your organizational anchor — who you are, what you are building, where you are headed. The first memory every organization builds. Every subsequent memory inherits from it.",
    "ID20": "Why you exist, how you deliver, what you create, who you serve, and your core offering. The identity bedrock.",
    "ID21": "The problem you solve, why now, and what sets the product apart.",
    "ID22": "What brings a customer the first time, and what keeps them.",
    "ID23": "Whether what you value is what the market and its culture reward.",
    "MA20": "Where you sit in the market, and the ground you hold that nobody else does.",
    "MA21": "How you create, deliver, and capture value.",
    "MA22": "How revenue is produced, and which lever grows it.",
    "IN20": "What ships now, next, and later, and the customer value each release carries.",
    "IN21": "How you scale from entry to dominance, and what each stage depends on.",
    "BR20": "What the brand stands for, and how it is meant to be perceived.",
    "BR21": "How the brand shows up in market, campaign by campaign.",
    "BR22": "Who you speak to, and how you speak to each of them.",
    "LA20": "The numbers that prove your values are working.",
    "LA21": "The human signals that confirm organizational health.",
}


def init(directory: Path) -> list[Path]:
    """Write the First Fifteen as skeleton files. Existing files are never overwritten."""
    directory.mkdir(parents=True, exist_ok=True)
    written: list[Path] = []
    for address, title in FIRST_FIFTEEN.items():
        path = directory / f"{address}.md"
        if path.exists():
            continue
        path.write_text(_SKELETON.format(address=address, title=title, about=_ABOUT[address]), encoding="utf-8")
        written.append(path)
    return written


# ----------------------------------------------------------------------------- cli


def _main(argv: list[str] | None = None) -> int:
    ap = argparse.ArgumentParser(prog="uca", description=__doc__.splitlines()[1].strip())
    ap.add_argument("addresses", nargs="*", help="addresses to parse (or 'init DIR')")
    ap.add_argument("--check", metavar="FILE", help="validate a file of addresses, one per line")
    ap.add_argument("--json", action="store_true", help="emit JSON")
    ap.add_argument("--allow-illustrative", action="store_true", help="admit Personal/Marketplace prefixes")
    ap.add_argument("--version", action="version", version=f"uca {__version__} (UCA 1.0)")
    args = ap.parse_args(argv)
    kw = {"allow_illustrative": args.allow_illustrative}

    if args.addresses[:1] == ["init"]:
        if len(args.addresses) != 2:
            ap.error("usage: uca init DIR")
        written = init(Path(args.addresses[1]))
        for p in written:
            print(f"wrote {p}")
        print(f"{len(written)} file(s) written; {15 - len(written)} already present")
        return 0

    if args.check:
        good, bad = check(Path(args.check).read_text(encoding="utf-8").splitlines(), **kw)
        if args.json:
            print(json.dumps({"valid": [a.to_dict() for a in good], "invalid": bad}, indent=2))
        else:
            for a in good:
                print(f"ok    {a.text:<28} {a.explain()}")
            for text, reason in bad:
                print(f"FAIL  {text:<28} {reason}")
            print(f"\n{len(good)} valid, {len(bad)} invalid")
        return 1 if bad else 0

    if not args.addresses:
        ap.print_help()
        return 2

    rc = 0
    for text in args.addresses:
        try:
            a = parse(text, **kw)
            print(json.dumps(a.to_dict(), indent=2) if args.json else f"ok    {a.text:<28} {a.explain()}")
        except UCAError as e:
            rc = 1
            print(f"FAIL  {e}", file=sys.stderr)
    return rc


if __name__ == "__main__":
    sys.exit(_main())
