#!/usr/bin/env python3
"""agentlore - git-reviewable repo memory for coding agents.

Typed memories live as markdown in .memory/ and are versioned with the code, so
every write shows up as a reviewable diff in the PR that motivated it.

Design rules baked in:
  * markdown is the source of truth; the SQLite index is DERIVED and gitignored
  * every memory is anchored to file globs, so code drift makes it STALE
  * memory is a hint, never ground truth -- check reports, it does not enforce
"""
from __future__ import annotations

import argparse
import fnmatch
import hashlib
import os
import re
import sqlite3
import subprocess
import sys
import uuid
from datetime import date, datetime
from pathlib import Path

VERSION = "0.9.0"

# AGENTS.md is read by every agent in every session, so the compiled dead-end list is
# bounded: past this many, the file points at .memory/dead-ends.md rather than growing.
COMPILE_DEAD_MAX = 12
MEM = Path(".memory")
INDEX = MEM / "index.db"
SOURCES = ("decisions.md", "dead-ends.md", "conventions.md", "hazards.md")
FILES = {"decision": "decisions.md", "dead-end": "dead-ends.md",
         "convention": "conventions.md", "hazard": "hazards.md"}
PREFIX = {"decision": "DEC", "dead-end": "DEAD", "convention": "CONV", "hazard": "HAZ"}
HEADING = {"decision": "Decisions", "dead-end": "Dead ends",
           "convention": "Conventions", "hazard": "Hazards"}
# Statuses that mean "history, not an obligation": excluded from live checks. A retracted
# memory was never true; a superseded one was true and has been replaced. Both stay on disk
# so a reviewer can see what was removed and why.
DEAD_STATUSES = ("superseded", "deprecated", "retracted")

FENCE = re.compile(r"<!--\s*agentlore\s*(.*?)-->", re.S)
GREEN, RED, DIM, BOLD, OFF = "\033[32m", "\033[31m", "\033[2m", "\033[1m", "\033[0m"


# ---------------------------------------------------------------- git helpers

def _run(cmd):
    try:
        p = subprocess.run(cmd, capture_output=True, text=True, check=False)
        return p.stdout.strip() if p.returncode == 0 else ""
    except Exception:
        return ""


def is_git():
    return _run(["git", "rev-parse", "--is-inside-work-tree"]) == "true"


def head_rev():
    return _run(["git", "rev-parse", "--short", "HEAD"])


def changed_since(rev):
    """Files changed between rev and HEAD, or None if we cannot tell."""
    if not rev or not is_git():
        return None
    if _run(["git", "cat-file", "-t", rev]) != "commit":
        return None
    out = _run(["git", "diff", "--name-only", "--relative", rev + "..HEAD"])
    if not out:
        # rev == HEAD, or the command failed: distinguish
        if _run(["git", "rev-parse", "--short", "HEAD"]) == rev:
            return []
        return None
    return [line for line in out.splitlines() if line.strip()]


def changed_set(since=None):
    """The files this change touches, or None when `since` cannot be resolved.

    Working tree by default (staged + unstaged + untracked), or relative to a ref
    when --since is given -- which is what the CI gate passes (the PR base sha).

    None is a SENTINEL, not an empty set. An unresolvable ref -- a sha a shallow clone
    never fetched, a typo, `github.event.before` on a new branch -- makes `git diff`
    produce nothing, and an empty change set silently disables dead-ends-settled. That is
    the same fail-open shape as trusting commit SHAs, so callers must treat None as an
    error rather than as "nothing changed".
    """
    if since:
        probe = _run(["git", "rev-parse", "--verify", "--quiet", since + "^{commit}"])
        if not probe.strip():
            return None
        for spec in (since + "...HEAD", since + "..HEAD"):
            out = _run(["git", "diff", "--name-only", "--relative", spec])
            files = set(l for l in out.splitlines() if l.strip())
            if files:
                return files
        return set()
    files = set()
    # --relative is load-bearing, not cosmetic. git reports paths relative to the REPO
    # ROOT; anchors and ".memory/<file>" are relative to the WORKING DIRECTORY, which is
    # the memory root. Without --relative the two never match whenever agentlore runs from a
    # subdirectory, so dead-ends-settled matches nothing and reports GREEN while checking
    # nothing -- a monorepo or service-subdirectory layout silently loses the gate.
    # (git ls-files --others is already cwd-relative, which is how this inconsistency hid.)
    for cmd in (["git", "diff", "--name-only", "--relative"],
                ["git", "diff", "--cached", "--name-only", "--relative"],
                ["git", "ls-files", "--others", "--exclude-standard"]):
        out = _run(cmd)
        files.update(l for l in out.splitlines() if l.strip())
    return files


def repo_name():
    if is_git():
        url = _run(["git", "remote", "get-url", "origin"])
        if url:
            return url.rstrip("/").split("/")[-1].replace(".git", "")
        return Path.cwd().name
    return Path.cwd().name


# ------------------------------------------------------------------- parsing

def parse_file(path):
    """Return the list of entries in one markdown file, each with its line number
    (line numbers matter: CI annotations must point at the entry in the diff)."""
    if not path.exists():
        return []
    text = path.read_text()
    marks = list(re.finditer(r"(?m)^##[ \t]+", text))
    entries = []
    for i, mark in enumerate(marks):
        stop = marks[i + 1].start() if i + 1 < len(marks) else len(text)
        chunk = text[mark.end():stop]
        nl = chunk.find("\n")
        title = (chunk[:nl] if nl >= 0 else chunk).strip()
        rest = chunk[nl + 1:] if nl >= 0 else ""
        metacomment = {}
        m = FENCE.search(rest)
        if m:
            for line in m.group(1).strip().splitlines():
                line = line.strip()
                if not line or line.startswith("#") or ":" not in line:
                    continue
                key, val = line.split(":", 1)
                metacomment[key.strip()] = val.strip()
            rest = rest[:m.start()] + rest[m.end():]
        e = dict(metacomment)
        e["title"] = title
        e["body"] = rest.strip()
        e["file"] = path.name
        e["lineno"] = text.count("\n", 0, mark.start()) + 1
        for key in ("anchors", "supersedes", "coexists_with"):
            e[key] = [x for x in re.split(r"[,\s]+", e.get(key, "")) if x]
        e.setdefault("id", "")
        e.setdefault("type", "")
        e.setdefault("status", "accepted")
        # A claim makes a memory comparable with other memories. Two live memories that
        # declare the SAME key but DIFFERENT values contradict each other -- and that is
        # the only contradiction this tool can decide without natural-language inference.
        e.setdefault("key", "")
        e.setdefault("value", "")
        e.setdefault("coexists_why", "")
        entries.append(e)
    return entries


def load_all():
    return [e for name in SOURCES for e in parse_file(MEM / name)]


def new_entry_id(mtype):
    """A fresh entry id, never one already in use.

    The suffix is 4 random hex chars (16 bits), so collisions are rare but real -- one
    collided in CI on a two-entry fixture and failed `ids-unique`. Rare is not safe: a
    duplicate id makes verify/supersede/retract ambiguous, which for this tool is worse
    than an ugly id. Regenerate on collision, then widen rather than loop forever.

    Reading the ids in use is therefore not optional. If that read fails, uniqueness cannot
    be promised, and continuing with an empty `taken` set is the quiet way to hand out an
    id that already exists -- so this fails closed instead.
    """
    try:
        taken = {e["id"] for e in load_all()}
    except Exception as exc:
        raise SystemExit(
            "agentlore: refusing to mint an id -- could not read the ids already in use (%s).\n"
            "      A duplicate id would make verify/supersede/retract ambiguous, so this\n"
            "      fails closed. Fix the unreadable memory file above, then retry." % exc)
    for attempt in range(10):
        width = 4 if attempt < 5 else 8
        eid = "%s-%s-%s" % (PREFIX[mtype], date.today().isoformat(),
                            uuid.uuid4().hex[:width])
        if eid not in taken:
            return eid
    raise SystemExit("could not generate a unique id after 10 attempts")


def source_hash():
    h = hashlib.sha256()
    for name in SOURCES:
        p = MEM / name
        h.update(name.encode())
        h.update(p.read_bytes() if p.exists() else b"")
    return h.hexdigest()


# ------------------------------------------------------------------- anchors

def _id_in_block(block, entry_id):
    """True when this block is the entry with exactly this id.

    Anchored, never a substring. The original substring test meant `verify 2026` matched
    every memory created in 2026 -- and a re-stamp resets the staleness clock, so a loose
    match would silently certify memories nobody named.
    """
    return re.search(r"(?m)^id:\s*%s\s*$" % re.escape(entry_id), block) is not None


def resolve_id(needle, entries=None):
    """Resolve a full id or an unambiguous prefix to exactly one id.

    Returns (entry_id, candidates). entry_id is None both when nothing matched and when
    more than one thing matched, and the caller must refuse in either case rather than
    guess which memory the user meant.
    """
    entries = load_all() if entries is None else entries
    ids = [e["id"] for e in entries]
    if needle in ids:
        return needle, []
    hits = sorted(i for i in ids if i.startswith(needle))
    if len(hits) == 1:
        return hits[0], []
    return None, hits


# ---------------------------------------------------------------------------
# Sensitive-data scanning
#
# .memory/ is a bad place for a secret, worse than ordinary source. Memories ride the PR, so a
# token pasted into a note is committed, reviewed as prose, and then compiled into AGENTS.md --
# which every agent loads into its context. A leak here is delivered on purpose, repeatedly.
#
# Decidable only, and high precision on purpose: a false positive blocks a legitimate memory, and
# a guard that cries wolf gets switched off, which is strictly worse than no guard. So secrets are
# matched by construction (provider prefixes, PEM blocks, secret-looking assignment with an opaque
# value) rather than by entropy alone -- a bare entropy rule would fire on the commit SHAs this
# tool stores in every evidence field.
# ---------------------------------------------------------------------------

SECRET_PATTERNS = [
    ("a private key block", re.compile(r"-----BEGIN [A-Z ]*PRIVATE KEY-----")),
    ("an AWS access key id", re.compile(r"\b(?:AKIA|ASIA|ABIA|ACCA)[0-9A-Z]{16}\b")),
    ("a GitHub token", re.compile(r"\b(?:gh[pousr]_[A-Za-z0-9]{20,}|github_pat_[A-Za-z0-9_]{20,})\b")),
    ("a Slack token", re.compile(r"\bxox[baprs]-[A-Za-z0-9-]{10,}\b")),
    ("an OpenAI key", re.compile(r"\bsk-[A-Za-z0-9]{32,}\b")),
    ("an Anthropic key", re.compile(r"\bsk-ant-[A-Za-z0-9\-_]{20,}\b")),
    ("a Google API key", re.compile(r"\bAIza[0-9A-Za-z\-_]{30,}\b")),
    ("a Stripe key", re.compile(r"\b(?:sk|rk)_(?:live|test)_[A-Za-z0-9]{20,}\b")),
    ("a JWT", re.compile(r"\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\b")),
    ("a bearer token", re.compile(r"(?i)\bbearer\s+[A-Za-z0-9\-._~+/]{20,}=*")),
    # A secret-shaped NAME with an opaque VALUE. The value must be long: this is what keeps the
    # rule off prose like "the token must be set in .env" and off short placeholders.
    #
    # The name may be part of a larger identifier, because in practice it almost always is:
    # gateway_token, AWS_SECRET_ACCESS_KEY, db_password, SLACK_BOT_TOKEN. A plain \b before the
    # word does NOT work here -- `_` is a word character, so \btoken fails inside `gateway_token`
    # and every one of those names sails through. The word must simply not be preceded by a letter
    # or digit. The suffix matters for the same reason: in AWS_SECRET_ACCESS_KEY the word is
    # followed by more of the name before the `=`.
    ("a secret-looking assignment", re.compile(
        r"(?i)(?<![A-Za-z0-9])[A-Za-z0-9]*[_-]?"
        r"(?:password|passwd|pwd|secret|token|api[_-]?key|apikey|access[_-]?key|"
        r"auth[_-]?token|client[_-]?secret|private[_-]?key|passphrase)"
        r"[A-Za-z0-9_-]*\s*[:=]\s*[\"']?([A-Za-z0-9\-_./+=]{16,})")),
]

# Always a failure: a memory has no legitimate reason to carry one of these.
PII_HARD_PATTERNS = [
    ("a US social security number", re.compile(r"\b\d{3}-\d{2}-\d{4}\b")),
    ("a phone number", re.compile(
        r"(?<!\d)(?:\+\d{1,3}[ .-]?)?(?:\(\d{3}\)|\d{3})[ .-]\d{3}[ .-]\d{4}(?!\d)")),
]

# A notice on its own -- naming a person by email or handle is often exactly what an ownership
# note should do. It becomes a failure only in BULK, because one address is a citation and ten
# are a customer list.
PII_SOFT_PATTERNS = [
    ("an email address", re.compile(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b")),
]

PII_BULK_THRESHOLD = 3


def _luhn(number):
    digits = [int(c) for c in number if c.isdigit()]
    if not 13 <= len(digits) <= 19:
        return False
    total, parity = 0, len(digits) % 2
    for i, d in enumerate(digits):
        if i % 2 == parity:
            d *= 2
            if d > 9:
                d -= 9
        total += d
    return total % 10 == 0


CARD_SHAPED = re.compile(r"(?<!\d)(?:\d[ -]?){12,18}\d(?!\d)")


def mask_value(text):
    """Never echo a secret, even the one you just found.

    This output lands in CI logs and shell history. A short prefix is enough to locate the value
    and not enough to use it.
    """
    text = text.strip()
    if len(text) <= 6:
        return "*" * len(text)
    return text[:2] + "*" * min(len(text) - 2, 16)


DIRECTIVE_PATTERNS = [
    # A memory IS an instruction, so "this sounds like an instruction" cannot be the rule --
    # that would flag the point of the file. These match a narrow slice with no legitimate
    # reading, or a reading narrow enough to be worth a stated reason.
    ("tells the agent to conceal something from the people reviewing it", re.compile(
        r"(?i)\b(?:do not|don't|never)\s+(?:tell|mention|reveal|disclose|report)\b"
        r"[^.]{0,40}?\b(?:the\s+)?(?:user|human|developer|reviewer|anyone|them)\b")),
    ("tells the agent to hide something from the people reviewing it", re.compile(
        r"(?i)\b(?:hide|conceal)\b[^.]{0,30}?\bfrom\s+(?:the\s+)?"
        r"(?:user|human|developer|reviewer)\b")),
    ("tells the agent to ignore its instructions", re.compile(
        r"(?i)\b(?:ignore|disregard|forget|override)\b[^.]{0,30}?"
        r"\b(?:previous|prior|earlier|above|system)\b[^.]{0,20}?\b(?:instruction|prompt|rule|message)s?\b")),
    ("tells the agent to bypass this gate", re.compile(
        r"(?i)\b(?:skip|bypass|disable|ignore|avoid)\b[^.]{0,30}?"
        r"\b(?:memory-health|memory health|agentlore check|the gate|this check|these checks)\b")),
    ("instructs the agent about credential handling", re.compile(
        r"(?i)\b(?:read|use|print|echo|export|send|upload|post|curl)\b[^.]{0,30}?"
        r"\b(?:the\s+)?(?:credential|api[ _-]?key|token|password|secret|private key)\b")),
    ("tells the agent to weaken a platform control", re.compile(
        r"(?i)\b(?:edit|modify|change|remove|delete|weaken|relax|disable)\b[^.]{0,40}?"
        r"\b(?:codeowners|branch protection|required (?:review|check|status)|permissions?)\b")),
]


def scan_directives(text):
    """Narrow, decidable slice: (kind, lineno, excerpt) for memory aimed at the agent itself.

    This is NOT injection detection, and the distinction is the whole design. A `.memory/`
    entry is supposed to be an instruction -- "use httpx, never requests" is the product
    working. So the rule cannot be "looks like an instruction". It matches only directives
    that target the agent's own process, concealment, credentials, or a platform control,
    where the legitimate reading is absent or needs a stated reason. Everything subtler is
    left to the reviewer, which is where it belongs.
    """
    out = []
    for i, line in enumerate(text.splitlines(), 1):
        for kind, rx in DIRECTIVE_PATTERNS:
            for m in rx.finditer(line):
                out.append((kind, i, m.group(0).strip()[:70]))
    return out


def scan_sensitive(text):
    """Findings, strongest first: (severity, kind, lineno, masked).

    severity is 'secret', 'pii-hard' or 'pii-soft'.
    """
    out = []
    lines = text.splitlines()
    for i, line in enumerate(lines, 1):
        # finditer, not search: two secrets on one line are two findings. `search` reported only
        # the first, which undercounted both leaks and the bulk-PII rule that depends on the count.
        for kind, rx in SECRET_PATTERNS:
            for m in rx.finditer(line):
                val = m.group(1) if m.groups() else m.group(0)
                out.append(("secret", kind, i, mask_value(val)))
        for kind, rx in PII_HARD_PATTERNS:
            for m in rx.finditer(line):
                out.append(("pii-hard", kind, i, mask_value(m.group(0))))
        for cm in CARD_SHAPED.finditer(line):
            if _luhn(cm.group(0)):
                out.append(("pii-hard", "a payment card number", i, mask_value(cm.group(0))))
        for kind, rx in PII_SOFT_PATTERNS:
            for m in rx.finditer(line):
                out.append(("pii-soft", kind, i, mask_value(m.group(0))))

    # Bulk soft PII is not a citation any more, it is a dataset.
    soft = [f for f in out if f[0] == "pii-soft"]
    if len({f[3] for f in soft}) >= PII_BULK_THRESHOLD:
        out = [(("pii-hard" if f[0] == "pii-soft" else f[0]), f[1], f[2], f[3]) for f in out]
        out.append(("pii-hard", "a bulk list of personal addresses", soft[0][2],
                    "%d distinct addresses" % len({f[3] for f in soft})))
    return out


def describe_findings(findings):
    """One line per finding, masked. Shared by `add` (refuse) and `check` (report)."""
    for sev, kind, lineno, masked in findings:
        print("    line %s: %s  [%s]" % (lineno, kind, masked))


def refuse_id(needle, candidates):
    """One refusal for every verb, so no verb's failure mode is a guess."""
    if candidates:
        print("ambiguous id %r -- %d memories match:" % (needle, len(candidates)))
        for c in candidates:
            print("  %s" % c)
        print("  pass a longer id. agentlore will not guess which one you meant.")
    else:
        print("no memory with id %s" % needle)
    return 1


def anchor_matches(glob, path):
    g = glob.strip()
    if g.endswith("/**"):
        g = g[:-3]
    elif g.endswith("/*"):
        g = g[:-2]
    if any(ch in g for ch in "*?["):
        return fnmatch.fnmatch(path, g)
    g = g.rstrip("/")
    return path == g or path.startswith(g + "/")


def anchor_files(glob):
    """Every real FILE an anchor matches.

    Two traps: (1) a trailing '/**' makes pathlib.glob yield DIRECTORIES only, so
    matching on the glob result never sees files; (2) an empty leftover directory
    after a file is deleted must NOT count as the code still existing.
    """
    g = glob.strip().rstrip("/")
    for suffix in ("/**", "/*"):
        if g.endswith(suffix):
            g = g[:-len(suffix)]
            break
    try:
        # A root-level glob ("**", "*", ".") means the whole repo. It has to be handled
        # explicitly because pathlib.glob("**") yields DIRECTORIES only, so an anchor of
        # "**" silently matched nothing. The skip set matters too: a fingerprint that
        # included .git would churn on every commit, and one that included .memory would
        # make every memory write invalidate every repo-wide memory.
        if g in ("", ".", "*", "**"):
            skip = {".git", ".memory"}
            return sorted(x for x in Path(".").rglob("*")
                          if x.is_file() and not (skip & set(x.parts)))
        if any(ch in g for ch in "*?["):
            return sorted(x for x in Path(".").glob(g) if x.is_file())
        p = Path(g)
        if p.is_file():
            return [p]
        return sorted(x for x in p.rglob("*") if x.is_file()) if p.is_dir() else []
    except Exception:
        return []


def anchor_has_files(glob):
    return bool(anchor_files(glob))


# Line-comment leader per extension, for the formatting-insensitive view only.
COMMENT_LEADER = {
    ".py": "#", ".rb": "#", ".sh": "#", ".bash": "#", ".zsh": "#", ".pl": "#",
    ".yaml": "#", ".yml": "#", ".toml": "#", ".ini": "#", ".cfg": "#",
    ".js": "//", ".jsx": "//", ".ts": "//", ".tsx": "//", ".go": "//", ".rs": "//",
    ".java": "//", ".c": "//", ".h": "//", ".cpp": "//", ".cc": "//", ".cs": "//",
    ".swift": "//", ".kt": "//", ".scala": "//", ".php": "//",
}


def format_view(path):
    """A formatting-insensitive view of one file: comments, blank lines and
    whitespace runs removed.

    Deliberately NOT a parser and NOT the authority. The RAW content stays the
    authority; this view can only ever DOWNGRADE an already-detected change from
    "the code changed" to "the code was reformatted". So a bug here can never hide
    a real change -- it can only make the report less alarming, never silent.
    """
    try:
        text = path.read_text()
    except Exception:
        return ""
    leader = COMMENT_LEADER.get(path.suffix.lower())
    lines = []
    for line in text.splitlines():
        if leader:
            if line.lstrip().startswith(leader):
                continue
            line = line.split(leader, 1)[0]
        line = re.sub(r"\s+", " ", line).strip()
        if line:
            lines.append(line)
    return "\n".join(lines)


def anchor_fingerprints(entry):
    """(raw, formatting-insensitive) fingerprints of every anchored file."""
    raw, view = [], []
    for g in (entry.get("anchors") or []):
        for p in anchor_files(g):
            try:
                content = p.read_bytes()
            except Exception:
                content = b""
            raw.append("%s:%s" % (p.as_posix(), hashlib.sha256(content).hexdigest()[:16]))
            view.append("%s:%s" % (p.as_posix(),
                        hashlib.sha256(format_view(p).encode()).hexdigest()[:16]))
    if not raw:
        return "", ""
    return (hashlib.sha256("\n".join(sorted(raw)).encode()).hexdigest()[:16],
            hashlib.sha256("\n".join(sorted(view)).encode()).hexdigest()[:16])


def anchor_fingerprint(entry):
    """The raw content fingerprint: the authority for drift.

    Deliberately NOT based on commit SHAs. Squash-merging and rebasing -- the
    default on many repos -- rewrite history, so a stored commit becomes
    unreachable and the staleness check would silently stop working. A memory
    system must never fail open, so drift is decided by CONTENT.
    """
    return anchor_fingerprints(entry)[0]


# ------------------------------------------------------------- CODEOWNERS glue
# A hazard ("do not touch billing") is a POINTER to a control, never the control itself.
# Nothing in a markdown file stops an agent editing a directory: CODEOWNERS, branch
# protection and permissions do. So a hazard has to name the mechanism that enforces it,
# and when it names CODEOWNERS the gate checks that the rule actually exists and covers
# the anchored path. That is the whole difference between a policy and a wish.

CODEOWNERS_PATHS = (".github/CODEOWNERS", "CODEOWNERS", "docs/CODEOWNERS")


def anchor_base(glob):
    """The concrete path a glob anchor stands for, for CODEOWNERS lookups."""
    g = glob.strip().lstrip("./")
    for suffix in ("/**", "/*"):
        if g.endswith(suffix):
            return g[:-len(suffix)]
    if any(ch in g for ch in "*?["):
        return g.rsplit("/", 1)[0] if "/" in g else g
    return g


def codeowners_rules():
    """(source, [(pattern, owners)]) in file order -- later rules win."""
    for name in CODEOWNERS_PATHS:
        p = Path(name)
        if not p.is_file():
            continue
        rules = []
        for line in p.read_text().splitlines():
            line = line.split("#", 1)[0].strip()
            if not line:
                continue
            parts = line.split()
            if len(parts) >= 2:
                rules.append((parts[0], parts[1:]))
        return name, rules
    return None, []


def codeowners_match(pattern, path):
    """Best-effort match of a CODEOWNERS pattern against a repo path.

    Not a reimplementation of GitHub's matcher (which is gitignore-flavoured and has
    subtleties around `**` and anchoring). Deliberately errs toward finding a rule: a
    false "covered" is a softer failure than failing a repo that is in fact protected.
    """
    pat = pattern.strip()
    anchored = pat.startswith("/")
    if anchored:
        pat = pat[1:]
    path = path.lstrip("./")
    if pat in ("*", "**"):
        return True
    if pat.endswith("/"):
        base = pat.rstrip("/")
        return path == base or path.startswith(base + "/")
    if pat.endswith("/**"):
        base = pat[:-3].rstrip("/")
        return path == base or path.startswith(base + "/")
    if not anchored:
        for part in path.split("/"):
            if fnmatch.fnmatch(part, pat):
                return True
    return fnmatch.fnmatch(path, pat)


def codeowners_owners(path):
    """Owners CODEOWNERS assigns to path, or [] if no rule covers it."""
    _, rules = codeowners_rules()
    owners = []
    for pat, own in rules:
        if codeowners_match(pat, path):
            owners = own              # last match wins, per CODEOWNERS semantics
    return owners


def stale_hits(entry, changed):
    if changed is None:
        return None
    globs = entry.get("anchors") or []
    if not globs:
        return []
    hits = []
    for f in changed:
        if f.startswith(".memory/"):
            continue
        for g in globs:
            if anchor_matches(g, f):
                hits.append(f)
                break
    return hits


# --------------------------------------------------------------------- index

def cmd_index(args):
    MEM.mkdir(exist_ok=True)
    entries = load_all()
    if INDEX.exists():
        INDEX.unlink()
    con = sqlite3.connect(str(INDEX))
    con.executescript(
        "CREATE TABLE entries (id TEXT PRIMARY KEY, type TEXT, status TEXT, title TEXT,"
        " file TEXT, body TEXT, author TEXT, created TEXT, verified_at TEXT, review_by TEXT,"
        " anchors TEXT, supersedes TEXT, evidence TEXT);"
        "CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT);"
    )
    for e in entries:
        con.execute(
            "INSERT OR REPLACE INTO entries VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?)",
            (e["id"], e["type"], e["status"], e["title"], e["file"], e["body"],
             e.get("author", ""), e.get("created", ""), e.get("verified_at", ""),
             e.get("review_by", ""), " ".join(e["anchors"]),
             " ".join(e["supersedes"]), e.get("evidence", "")),
        )
    con.execute("INSERT INTO meta VALUES ('source_hash', ?)", (source_hash(),))
    con.execute("INSERT INTO meta VALUES ('built_at', ?)", (datetime.now().isoformat(timespec="seconds"),))
    con.commit()
    con.close()
    print("index built: %d memories -> %s" % (len(entries), INDEX))
    return 0


def cmd_init(args):
    MEM.mkdir(exist_ok=True)
    for name in SOURCES:
        p = MEM / name
        if not p.exists():
            kind = [k for k, v in FILES.items() if v == name][0]
            p.write_text("# %s\n\n<!-- agentlore: one entry per '## ' heading -->\n" % HEADING[kind])
    gi = Path(".gitignore")
    line = ".memory/index.db"
    existing = gi.read_text() if gi.exists() else ""
    if line not in existing:
        gi.write_text(existing.rstrip("\n") + ("\n" if existing else "") + line + "\n")
    print("initialised .memory/ (%s)" % ", ".join(SOURCES))
    print("  source of truth: .memory/*.md   derived+gitignored: .memory/index.db")
    return cmd_index(args)


# ----------------------------------------------------------------------- add

def cmd_add(args):
    if (args.key or "").strip() and not (args.value or "").strip():
        print("--key needs --value: a claim with no value cannot be compared with anything.")
        return 1
    if (args.value or "").strip() and not (args.key or "").strip():
        print("--value needs --key: name the thing this memory makes a claim about.")
        return 1
    if args.coexists_with and not (args.coexists_why or "").strip():
        print("--coexists-with needs --coexists-why.")
        print("  Declaring two contradictory memories coexistent silences a real conflict,")
        print("  so the declaration has to carry its reason. Nothing gets switched off")
        print("  anonymously -- same rule as retract.")
        return 1
    if args.allow_directive is not None and not args.allow_directive.strip():
        print("--allow-directive needs a reason: nothing gets switched off anonymously.")
        return 1
    MEM.mkdir(exist_ok=True)
    fname = FILES[args.type]
    path = MEM / fname
    if not path.exists():
        path.write_text("# %s\n" % HEADING[args.type])
    eid = new_entry_id(args.type)
    verified = head_rev() or ""
    lines = [
        "",
        "## %s%s" % (args.title, " (settled)" if args.resolved_by else ""),
        "<!-- agentlore",
        "id: %s" % eid,
        "type: %s" % args.type,
        "status: %s" % args.status,
    ]
    if args.anchor:
        lines.append("anchors: %s" % " ".join(args.anchor))
    if args.evidence:
        lines.append("evidence: %s" % args.evidence)
    if args.owner:
        lines.append("owner: %s" % args.owner)
    if args.enforcement:
        lines.append("enforcement: %s" % args.enforcement)
    if args.supersedes:
        lines.append("supersedes: %s" % " ".join(args.supersedes))
    if args.key:
        lines.append("key: %s" % args.key)
        lines.append("value: %s" % args.value)
    if args.coexists_with:
        lines.append("coexists_with: %s" % " ".join(args.coexists_with))
        lines.append("coexists_why: %s" % args.coexists_why)
    if args.allow_directive:
        lines.append("directive_reviewed: %s" % args.allow_directive.strip())
    raw_fp, norm_fp = anchor_fingerprints({"anchors": args.anchor})
    lines += [
        "author: %s" % args.author,
        "created: %s" % date.today().isoformat(),
        "verified_at: %s" % verified,
        "anchor_hash: %s" % raw_fp,
        "anchor_norm: %s" % norm_fp,
    ]
    if args.resolved_by:
        lines.append("resolved_by: %s" % args.resolved_by)
    if args.review_by:
        lines.append("review_by: %s" % args.review_by)
    lines += ["-->", "", args.body.strip(), ""]

    # Refuse BEFORE writing. A secret that reaches this file is committed, reviewed as prose, and
    # then compiled into AGENTS.md for every agent to load -- so the only cheap moment to stop it
    # is here, and a guard that warns and writes anyway has not guarded anything.
    candidate = "\n".join(lines)
    found = scan_sensitive(candidate)
    hard = [f for f in found if f[0] in ("secret", "pii-hard")]
    soft = [f for f in found if f[0] == "pii-soft"]
    if hard:
        print("refused: this memory looks like it carries a secret or personal data.")
        describe_findings(hard)
        print("  .memory/ rides the PR and is compiled into AGENTS.md, which every agent loads,")
        print("  so a value recorded here is committed, reviewed as prose and delivered on purpose.")
        print("  Nothing was written. Refer to the secret by name and keep the value in .env or")
        print("  the secret manager, e.g. \"the gateway key lives in the WARDEN_API_KEY env var\".")
        return 1
    if soft:
        print("note: this memory names personal data. Fine for an owner or an escalation path --")
        print("      but if this is a record of a person rather than a reference to one, stop.")
        describe_findings(soft)

    with path.open("a") as fh:
        fh.write("\n".join(lines))
    print("added %s (%s) -> %s" % (eid, args.type, fname))
    if not args.anchor:
        print("  %s! no --anchor: this memory can never go stale.%s" % (RED, OFF))
    if args.type == "dead-end" and not args.evidence:
        print("  %s! no --evidence: unsupported dead ends are superstition.%s" % (RED, OFF))
    if args.type == "hazard" and not args.owner:
        print("  %s! no --owner: a frozen area with nobody accountable is nobody's problem.%s"
              % (RED, OFF))
    if args.type == "hazard" and not args.enforcement:
        print("  %s! no --enforcement: this file cannot stop anyone editing anything. Name the "
              "control that actually enforces it (CODEOWNERS, branch protection, permissions), "
              "or it is a wish.%s" % (RED, OFF))
    if args.type == "hazard" and args.enforcement and "codeowners" in args.enforcement.lower():
        missing = [g for g in args.anchor if not codeowners_owners(anchor_base(g))]
        if missing:
            print("  %s! enforcement names CODEOWNERS, but no rule covers: %s%s"
                  % (RED, ", ".join(missing), OFF))
    return cmd_index(args)


# ------------------------------------------------------------------ CI output

def gh_escape(text):
    return text.replace("%", "%25").replace("\r", "%0D").replace("\n", "%0A")


def collect_findings(entries, live, index_ok, index_fresh, stale, restyled=None, unsettled=None,
                     hazards=None, contradictions=None, coexisting=None, unvalued=None,
                     chains=None, unresolved_since=""):
    """Turn every problem into a CI annotation anchored at the offending line."""
    findings = []
    known = set(e["id"] for e in entries)

    def F(level, entry, title, message):
        findings.append({
            "level": level, "title": title, "message": message,
            "file": (".memory/" + entry["file"]) if entry else "",
            "line": entry["lineno"] if entry else "",
        })

    for e, hits in stale:
        F("error", e, "MEMORY-HEALTH: stale memory",
          '%s "%s" is out of date: %s changed since it was last verified (%s). Resolve it in this PR with '
          '"agentlore verify %s" (still true) or "agentlore supersede %s <new-id>" (no longer true).'
          % (e["id"], e["title"], ", ".join(hits[:4]) or "its anchored files",
             e.get("verified_at") or "?", e["id"], e["id"]))
        # The memory file is usually NOT part of this PR's diff, so an annotation on it
        # would never render inline. Put a second one on the code that invalidated it.
        for path in hits[:1]:
            findings.append({
                "level": "error", "file": path, "line": "",
                "title": "MEMORY-HEALTH: this change invalidates a memory",
                "message": 'This change makes %s "%s" (in .memory/%s) out of date. '
                           'Resolve it in this PR: "agentlore verify %s" if it is still true, '
                           'or "agentlore supersede %s <new-id>" if not, then "agentlore index".'
                           % (e["id"], e["title"], e["file"], e["id"], e["id"]),
            })
    for e in live:
        if not (e.get("anchor_hash") or "").strip():
            F("error", e, "MEMORY-HEALTH: memory not stamped",
              "%s has no anchor fingerprint, so code drift cannot be detected for it. "
              "Run: agentlore verify %s" % (e["id"], e["id"]))
        for g in e["anchors"]:
            if not anchor_has_files(g):
                F("error", e, "MEMORY-HEALTH: anchor has no code",
                  "%s is anchored to %s, which no longer exists." % (e["id"], g))
        if e["type"] == "dead-end" and not e.get("evidence"):
            F("error", e, "MEMORY-HEALTH: unsupported dead end",
              '%s "%s" has no evidence. Add --evidence with the commit, PR or error that proved it.'
              % (e["id"], e["title"]))
        rb = e.get("review_by", "")
        if rb and rb < date.today().isoformat():
            F("error", e, "MEMORY-HEALTH: review overdue", "%s was due for review on %s." % (e["id"], rb))
        for ref in e["supersedes"]:
            if ref not in known:
                F("error", e, "MEMORY-HEALTH: broken supersession",
                  "%s claims to supersede %s, which does not exist." % (e["id"], ref))
    for e, touched in (unsettled or []):
        F("error", e, "MEMORY-HEALTH: dead end rode its own fix",
          '"%s" is anchored to %s, which this same change also modified -- so the memory '
          'describes the world BEFORE the change and now reads as a live warning about a '
          'condition your change removed. State whether it still holds: '
          'agentlore verify %s --resolved-by "<commit or PR that settled it>". If it is really '
          'settled, retype it as a decision and write the title about the CURRENT rule.'
          % (e["title"], ", ".join(touched), e["id"]))
        for path in touched[:1]:
            findings.append({
                "level": "error", "file": path, "line": "",
                "title": "MEMORY-HEALTH: this change may have settled a dead end",
                "message": "This file settled the dead end recorded as %s. Make sure that "
                           "memory now reads as history, not as a live warning." % e["id"],
            })
    for e in (restyled or []):
        F("notice", e, "MEMORY-HEALTH: reformatted, content equivalent",
          "%s is anchored to files that were reformatted or re-commented. The code did not "
          "meaningfully change, so this is not a failure. Run agentlore verify %s to clear it."
          % (e["id"], e["id"]))
    for e, why in (hazards or []):
        F("error", e, "MEMORY-HEALTH: unbacked hazard",
          '"%s" freezes %s but nothing enforces it (%s). A memory cannot stop an agent '
          'editing a directory -- CODEOWNERS, branch protection or permissions do. Either '
          'name the control that really enforces it with --enforcement, or add the '
          'CODEOWNERS rule it claims, or drop the hazard: a wish is worse than nothing.'
          % (e["title"], ", ".join(e["anchors"]) or "nothing", why))
    if unresolved_since:
        F("error", None, "MEMORY-HEALTH: --since cannot be resolved",
          'The change boundary "%s" does not resolve in this clone, so the check that catches '
          'a dead end recorded by the change that settled it is BLIND. This fails the build on '
          'purpose: an unresolvable boundary yields an empty change set, and reporting "blind" '
          'as "fine" is the fail-open shape this tool exists to prevent. Fix: fetch the base '
          'commit (actions/checkout with fetch-depth: 0), pass a sha that exists, or drop '
          '--since if you do not need that check.' % unresolved_since)
    for a, b in (contradictions or []):
        key = a.get("key", "")
        F("error", b, "MEMORY-HEALTH: two memories contradict each other",
          'Two live memories claim different things about "%s": %s says "%s", %s (in '
          '.memory/%s) says "%s". agentlore will not pick a winner -- both are surfaced, and '
          'resolving it is a human act. If %s is now wrong: "agentlore supersede %s %s". If it '
          'was never true: "agentlore retract %s --reason ...". If both hold in different '
          'contexts: re-add %s with --coexists-with %s --coexists-why "...". Then agentlore index.'
          % (key, a["id"], a["value"], b["id"], b["file"], b["value"],
             a["id"], a["id"], b["id"], a["id"], b["id"], a["id"]))
    for e, msg in (chains or []):
        F("error", e, "MEMORY-HEALTH: broken retirement chain",
          "The chain of replacements is broken: " + msg)
    for a, b in (coexisting or []):
        F("notice", b, "MEMORY-HEALTH: coexisting claims (declared)",
          'These two disagree about "%s" and that is recorded as intentional: %s'
          % (a.get("key", ""), a.get("coexists_why") or b.get("coexists_why") or "(no reason given)"))
    for e in (unvalued or []):
        F("notice", e, "MEMORY-HEALTH: claim without a value",
          '%s declares key "%s" but no value, so nothing can be compared with it. '
          'Add --value, or drop --key.' % (e["id"], e.get("key", "")))
    if not index_ok:
        F("error", None, "MEMORY-HEALTH: no index", "The derived index is missing. Run: agentlore index")
    elif not index_fresh:
        F("error", None, "MEMORY-HEALTH: index out of date",
          "The markdown was edited without rebuilding the index. Run: agentlore index")
    return findings


def emit_github(findings, passed, total, n_entries, n_stale):
    import os
    for f in findings:
        if f["file"] and f["line"]:
            loc = "file=%s,line=%s," % (f["file"], f["line"])
        elif f["file"]:
            # file-level annotation: renders on the PR even with no line in the hunk
            loc = "file=%s," % f["file"]
        else:
            loc = ""
        print("::%s %stitle=%s::%s" % (f["level"], loc, f["title"], gh_escape(f["message"])))
    path = os.environ.get("GITHUB_STEP_SUMMARY")
    if not path:
        return
    icon = "PASS" if passed == total else "FAIL"
    out = ["## MEMORY-HEALTH %s: %d/%d" % (icon, passed, total), "",
           "`%d memories, %d stale, %d findings`" % (n_entries, n_stale, len(findings)), ""]
    if findings:
        out += ["| where | what |", "|---|---|"]
        for f in findings:
            where = ("`%s:%s`" % (f["file"], f["line"])) if f["file"] else "repo-wide"
            out.append("| %s | %s |" % (where, f["message"]))
    else:
        out.append("All memory checks pass. Memory is a hint; code and failing tests win.")
    out += ["", "_Resolve a stale memory in this PR: `agentlore verify <id>` (still true) "
            "or `agentlore supersede <id> <new-id>` (no longer true). Then `agentlore index`._"]
    try:
        with open(path, "a") as fh:
            fh.write("\n".join(out) + "\n")
    except Exception:
        pass


# --------------------------------------------------------------------- check

def cmd_check(args):
    if not MEM.exists():
        print("MEMORY-HEALTH: 0/1 %sRED%s - no .memory/ in %s (run: agentlore init)" % (RED, OFF, Path.cwd()))
        return 1

    checks = []
    notes = []

    def add(name, ok, note=""):
        checks.append((name, ok, note))

    entries = load_all()

    # 1 index present
    index_ok = INDEX.exists()
    add("index-present", index_ok, "run: agentlore index")

    # 2 index fresh
    fresh = False
    detail = "n/a (no index)"
    if index_ok:
        try:
            con = sqlite3.connect(str(INDEX))
            row = con.execute("SELECT value FROM meta WHERE key='source_hash'").fetchone()
            con.close()
            have = row[0] if row else ""
            fresh = have == source_hash()
            detail = "run: agentlore index" if not fresh else ""
        except Exception as exc:
            detail = "index unreadable (%s)" % exc
    add("index-fresh", fresh, detail)

    # 3 entries parse / have ids
    missing = [e["title"] for e in entries if not e["id"]]
    add("ids-present", not missing, "no id: %s" % ", ".join(missing[:3]))

    # 4 ids unique
    seen, dups = set(), []
    for e in entries:
        if e["id"] in seen:
            dups.append(e["id"])
        seen.add(e["id"])
    add("ids-unique", not dups, "duplicate: %s" % ", ".join(dups[:3]))

    # 5 supersedes resolve
    known = set(e["id"] for e in entries) | set(e["id"] for e in entries if e["id"])
    dangling = []
    for e in entries:
        for ref in e["supersedes"]:
            if ref not in known:
                dangling.append("%s -> %s" % (e["id"], ref))
    add("supersedes-resolve", not dangling, "; ".join(dangling[:3]))

    # retired memories are history, not obligations: only live entries are checked
    live = [e for e in entries if e["status"] not in DEAD_STATUSES]

    # 6 anchors resolve to real files
    bad_anchor = []
    for e in live:
        for g in e["anchors"]:
            if not anchor_has_files(g):
                bad_anchor.append("%s anchors %s (no such file)" % (e["id"], g))
    add("anchors-resolve", not bad_anchor, "; ".join(bad_anchor[:3]))

    # 7 dead ends carry evidence
    unevidenced = [e["id"] for e in live if e["type"] == "dead-end" and not e.get("evidence")]
    add("dead-ends-evidenced", not unevidenced, ", ".join(unevidenced[:3]))

    # 8 review_by not expired
    expired = []
    today = date.today().isoformat()
    for e in live:
        rb = e.get("review_by", "")
        if rb and rb < today:
            expired.append("%s (review_by %s)" % (e["id"], rb))
    add("reviews-current", not expired, "; ".join(expired[:3]))

    # 9 anchors have not drifted since they were stamped. Decided by CONTENT, not by
    # git history: a squashed or rebased commit must never be able to disable this.
    unverifiable, stale, restyled = [], [], []
    for e in live:
        stored = (e.get("anchor_hash") or "").strip()
        if not stored:
            unverifiable.append(e["id"])
            continue
        raw_fp, norm_fp = anchor_fingerprints(e)
        if raw_fp == stored:
            continue
        stored_norm = (e.get("anchor_norm") or "").strip()
        if stored_norm and norm_fp == stored_norm:
            # Bytes differ, meaning does not: reformatted. Still reported -- never
            # silent -- but it is not a lie, so it does not break the build.
            restyled.append(e)
            continue
        hits = stale_hits(e, changed_since(e.get("verified_at", "")))
        stale.append((e, sorted(set(hits or []))))
    add("anchors-not-stale", not stale, "%d out of date" % len(stale))

    # 10 every live memory is stamped, so drift is detectable at all (fail closed)
    add("anchors-stamped", not unverifiable, "unstamped: %s" % ", ".join(unverifiable[:3]))

    # 11 a dead end may not ride the change that settled it. This catches the failure
    # observed in practice: an agent fixes a problem and records the problem as if it
    # were still live, so the memory warns about a condition its own change removed.
    changed = changed_set(args.since)
    # Fail closed. If the boundary cannot be resolved this check is blind, and "blind"
    # must never be reported as "fine": a sha that a shallow clone never fetched would
    # otherwise disable this check silently, which is the pitfall the whole design exists
    # to avoid.
    unresolved_since = args.since if changed is None else ""
    changed = changed or set()
    unsettled = []
    for e in live:
        if e["type"] != "dead-end" or (e.get("resolved_by") or "").strip():
            continue
        if (".memory/" + e["file"]) not in changed:
            continue
        touched = sorted(f for f in changed
                         if f != ".memory/" + e["file"]
                         and any(anchor_matches(g, f) for g in e["anchors"]))
        if touched:
            unsettled.append((e, touched[:3]))
    add("dead-ends-settled", not unsettled and not unresolved_since,
        ("--since %s does not resolve in this clone, so this check cannot see what changed"
         % unresolved_since) if unresolved_since else "%d unsettled" % len(unsettled))

    # 12 a hazard must name a real control and an accountable owner. A prohibition that no
    # mechanism enforces is a wish, and a wish is the worst kind of memory: an agent obeys
    # it and refuses legitimate work, or ignores it and ships an incident. When the hazard
    # names CODEOWNERS, verify the rule actually covers the anchored path -- so the memory
    # and the control cannot drift apart, which is the only way a soft store can carry a
    # hard policy.
    hazards_unbacked = []
    for e in live:
        if e["type"] != "hazard":
            continue
        why = []
        if not (e.get("owner") or "").strip():
            why.append("no owner")
        enf = (e.get("enforcement") or "").strip()
        if not enf:
            why.append("no enforcement named")
        elif "codeowners" in enf.lower():
            for g in e["anchors"]:
                owners = codeowners_owners(anchor_base(g))
                if not owners:
                    why.append("CODEOWNERS has no rule for %s" % g)
                    continue
                # If the hazard names an owner as a CODEOWNERS-style handle, the control must
                # agree with it. Otherwise a memory can claim "owned by @epic/security" while
                # CODEOWNERS puts the path under a catch-all -- the claim is unbacked even
                # though the path is nominally covered.
                claimed = (e.get("owner") or "").strip()
                if "@" in claimed or "/" in claimed:
                    norm = lambda s: s.lstrip("@").lower()
                    if norm(claimed) not in [norm(o) for o in owners]:
                        why.append("%s: CODEOWNERS assigns %s, hazard claims %s"
                                   % (g, " ".join(owners), claimed))
        if why:
            hazards_unbacked.append((e, "; ".join(why)))
    add("hazards-enforced", not hazards_unbacked, "%d unbacked" % len(hazards_unbacked))

    # 13 two live memories that claim the same thing must not disagree about it. This is
    # the only contradiction that can be decided WITHOUT natural-language inference: the
    # author declared both memories talk about the same key, so a differing value is a
    # contradiction by construction rather than a guess. Prose that contradicts other prose
    # is deliberately never guessed at -- similarity is not contradiction ("David works at
    # Google" and "Sarah works at Microsoft" score as similar), and the systems that do
    # guess retire true memories silently. See README.
    claims = {}
    for e in live:
        k = (e.get("key") or "").strip()
        if k and (e.get("value") or "").strip():
            claims.setdefault(k, []).append(e)
    contradictions, coexisting = [], []
    for key, group in sorted(claims.items()):
        for i, a in enumerate(group):
            for b in group[i + 1:]:
                if a["value"].strip() == b["value"].strip():
                    continue
                # An explicit, reasoned coexistence declaration means a human already made
                # the call. A genuinely context-dependent pair must be preserved, not
                # collapsed: forcing a winner would destroy a true memory.
                if b["id"] in a["coexists_with"] or a["id"] in b["coexists_with"]:
                    coexisting.append((a, b))
                else:
                    contradictions.append((a, b))
    unvalued = [e for e in live
                if (e.get("key") or "").strip() and not (e.get("value") or "").strip()]
    add("claims-agree", not contradictions, "%d contradiction(s)" % len(contradictions))

    # 14 following a retirement pointer must terminate at something live. A cycle, or a
    # chain landing on a superseded or retracted memory, means the reviewer who follows
    # "this was replaced by X" arrives at a dead end: the area has no live rule and nothing
    # says so. Hand-editing and retract-after-supersede are the usual routes in.
    byid = {e["id"]: e for e in entries if e["id"]}
    chain_problems = []
    replaced_by = {}
    for e in entries:
        nxt = (e.get("superseded_by") or "").strip()
        if not nxt:
            continue
        if nxt in byid:
            replaced_by[e["id"]] = nxt
        else:
            # A retirement pointing at nothing leaves the reader with no live rule AND no
            # signal that anything is wrong -- the same dead end as a cycle, reached faster.
            chain_problems.append((e, "%s is marked as replaced by %s, but no memory with that "
                                      "id exists, so the retirement points nowhere. Point it at "
                                      "the real successor, or re-stamp it if it is still true "
                                      "(agentlore verify %s)." % (e["id"], nxt, e["id"])))
    for e in entries:                      # "I supersede X" is the same edge, read forwards
        for ref in e["supersedes"]:
            if ref in byid:
                replaced_by.setdefault(ref, e["id"])
    for start in sorted(replaced_by):
        seen, cur, cycle = [start], replaced_by[start], False
        while cur in replaced_by:
            if cur in seen:
                cycle = True
                # Report each cycle once, from its lowest-sorted member, and show the loop
                # itself rather than a self-referential sentence.
                loop = seen[seen.index(cur):]
                if start == min(loop):
                    chain_problems.append((byid[start],
                                           "%s -- a cycle, so the chain never reaches a live "
                                           "memory and the area has no rule at all. Break it by "
                                           "re-stamping whichever entry is still true "
                                           "(agentlore verify <id>) or retracting it "
                                           "(agentlore retract <id> --reason ...)."
                                           % " -> ".join(loop + [cur])))
                break
            seen.append(cur)
            cur = replaced_by[cur]
        if cycle:
            continue
        end = byid.get(cur)
        if end is not None and end["status"] in DEAD_STATUSES:
            chain_problems.append((byid[start],
                                   "%s is replaced by %s, but %s is %s -- so following the "
                                   "pointer leads to a memory that is itself retired. Point it at "
                                   "the live successor instead."
                                   % (start, cur, cur, end["status"])))
    add("supersedes-acyclic", not chain_problems, "%d broken chain(s)" % len(chain_problems))

    # 15 the committed block is what compile would produce right now
    #
    # A memory that is not compiled into the file agents actually read is a memory nothing
    # delivers, and the failure is silent from every direction: the gate goes green, the
    # agent gets on with its work, and the entry simply never arrives. This check exists
    # because that happened: a new entry type began being compiled, the committed AGENTS.md
    # was never regenerated, and consumers ran on a stale file until CI caught it.
    agent_md = Path("AGENTS.md")
    if not agent_md.exists():
        add("compile-current", True, "no AGENTS.md here (not compiled)")
    elif "<!-- agentlore:begin" not in agent_md.read_text():
        add("compile-current", True, "AGENTS.md has no compiled block")
    else:
        committed = re.search(r"<!-- agentlore:begin.*?<!-- agentlore:end -->", agent_md.read_text(), re.S)
        add("compile-current", bool(committed) and committed.group(0) == compile_block(),
            "AGENTS.md is stale; run: agentlore compile")

    # 16/17 nothing in .memory/ should be a secret or a record of a person
    #
    # `add` refuses these at the door, but a memory can also arrive by hand-edit or in a
    # contributor's PR, which is exactly how a leak reaches a repo without anyone running the
    # tool. So the check re-reads the files rather than trusting the write path.
    secret_hits, pii_hits, soft_hits = [], [], []
    for mf in sorted(MEM.glob("*.md")):
        for sev, kind, lineno, masked in scan_sensitive(mf.read_text()):
            row = ("%s:%s" % (mf.name, lineno), kind, masked)
            if sev == "secret":
                secret_hits.append(row)
            elif sev == "pii-hard":
                pii_hits.append(row)
            else:
                soft_hits.append(row)
    add("no-secrets", not secret_hits,
        "possible secret in .memory/: %s" % ", ".join(r[0] for r in secret_hits[:3]))
    add("no-pii", not pii_hits,
        "possible personal data in .memory/: %s" % ", ".join(r[0] for r in pii_hits[:3]))

    # `.memory/` is instructions to a machine that holds credentials, compiled into AGENTS.md
    # where every session reads it. What can be decided is the narrow slice above; the rest is
    # a review problem, and this tool's answer to that is that memory rides the PR.
    # Scan the FILE, not a rebuilt entry: parse_file strips the metacomment fence and trims
    # blank lines, so rebuilding "title + body" shifts every line number and the annotation
    # lands in the wrong place. A hit is attributed to the entry whose block contains it, so a
    # stated exemption is honoured at the entry it was recorded on.
    directive_hits = []
    for mf in sorted(MEM.glob("*.md")):
        hits = scan_directives(mf.read_text())
        if not hits:
            continue
        blocks = sorted((e for e in entries if e.get("file") == mf.name),
                        key=lambda e: int(e.get("lineno") or 0))
        for kind, lineno, excerpt in hits:
            owner = None
            for i, b in enumerate(blocks):
                start = int(b.get("lineno") or 0)
                stop = (int(blocks[i + 1].get("lineno") or 0) - 1
                        if i + 1 < len(blocks) else sys.maxsize)
                if start <= lineno <= stop:
                    owner = b
                    break
            if owner is not None and (owner.get("directive_reviewed") or "").strip():
                continue
            directive_hits.append(("%s:%s" % (mf.name, lineno), kind, excerpt))
    add("no-agent-directives", not directive_hits,
        "memory instructs the agent about its own process: %s" % ", ".join(r[0] for r in directive_hits[:3]))

    passed = sum(1 for _, ok, _ in checks if ok)
    total = len(checks)
    color = GREEN if passed == total else RED

    findings = collect_findings(entries, live, index_ok, fresh, stale, restyled, unsettled,
                                hazards_unbacked, contradictions, coexisting, unvalued,
                                chain_problems, unresolved_since)
    # Never echo the value -- these annotations land in CI logs and are readable by anyone who
    # can read the repo. Type, file and line are enough to find it; the value is not needed.
    for rows, level, title in ((secret_hits, "error", "MEMORY-HEALTH: a secret may be recorded in the memory"),
                               (pii_hits, "error", "MEMORY-HEALTH: personal data may be recorded in the memory"),
                               (soft_hits, "notice", "MEMORY-HEALTH: personal data named in the memory")):
        for where, kind, masked in rows:
            fname, _, lineno = where.partition(":")
            findings.append({
                "level": level, "file": ".memory/" + fname, "line": lineno, "title": title,
                "message": "This looks like %s [%s]. .memory/ rides the PR and is compiled into "
                           "AGENTS.md, which every agent loads, so a value recorded here is "
                           "committed and delivered on purpose. Refer to it by name and keep the "
                           "value in .env or the secret manager." % (kind, masked),
            })

    # Directives get their own wording: the secret message ("keep the value in .env") is the
    # wrong advice for an instruction.
    for where, kind, excerpt in directive_hits:
        fname, _, lineno = where.partition(":")
        findings.append({
            "level": "error", "file": ".memory/" + fname, "line": lineno,
            "title": "MEMORY-HEALTH: the memory instructs the agent about its own process",
            "message": "This %s: %s -- and .memory/ is compiled into AGENTS.md, which every "
                       "agent loads every session, so an instruction here changes how it "
                       "behaves from then on. If it is deliberate, record why: re-add the "
                       "entry with --allow-directive \"reason\"." % (kind, excerpt),
        })

    if args.brief:
        bad = [n for n, ok, _ in checks if not ok]
        tail = "" if not bad else " -- " + ", ".join(bad)
        print("MEMORY-HEALTH: %d/%d %s%s" % (passed, total, "GREEN" if passed == total else "BROKEN", tail))
        if args.github:
            emit_github(findings, passed, total, len(entries), len(stale))
        return 0 if passed == total else 1

    print("%sMEMORY-HEALTH: %d/%d %s%s" % (color, passed, total,
          "GREEN" if passed == total else "BROKEN", OFF)
          + "  %s(%d memories, %d stale, %s)%s" % (
              DIM, len(entries), len(stale), repo_name(), OFF))
    for name, ok, note in checks:
        mark = GREEN + "ok  " + OFF if ok else RED + "FAIL" + OFF
        print("  %s %s%s" % (mark, name, ("  -- " + note) if note and not ok else ""))
    if stale:
        print("\n%sstale memories (code moved, memory did not):%s" % (BOLD, OFF))
        for e, hits in stale:
            print("  %s%s%s  %s" % (BOLD, e["id"], OFF, e["title"]))
            print("    anchored to %s" % (", ".join(e["anchors"]) or "-"))
            print("    changed since %s: %s" % (e.get("verified_at") or "-",
                  ", ".join(hits[:4]) or "anchor content changed"))
            print("    %s-> re-verify, or run: agentlore supersede %s%s" % (DIM, e["id"], OFF))
    if unverifiable:
        print("\n%s%d memories have no anchor fingerprint, so drift cannot be detected: %s%s"
              % (DIM, len(unverifiable), ", ".join(unverifiable[:5]), OFF))
        print("%s-> stamp them: agentlore verify <id>%s" % (DIM, OFF))
    if unsettled:
        print("\n%sdead ends that rode the change that settled them:%s" % (BOLD, OFF))
        for e, touched in unsettled:
            print("  %s%s%s  %s" % (BOLD, e["id"], OFF, e["title"]))
            print("    anchored to %s, changed in this same change: %s"
                  % (", ".join(e["anchors"]), ", ".join(touched)))
            print("    %s-> this memory now reads as a live warning about a condition your"
                  " change removed.%s" % (DIM, OFF))
            print("    %s-> state it: agentlore verify %s --resolved-by \"<what settled it>\"%s"
                  % (DIM, e["id"], OFF))
            print("    %s-> or retype it as a decision and write the title about the CURRENT"
                  " rule.%s" % (DIM, OFF))
    if hazards_unbacked:
        print("\n%sunbacked hazards (a frozen area nothing actually enforces):%s" % (BOLD, OFF))
        for e, why in hazards_unbacked:
            print("  %s%s%s  %s" % (BOLD, e["id"], OFF, e["title"]))
            print("    claims: %s" % (", ".join(e["anchors"]) or "-"))
            print("    problem: %s" % why)
            print("    %s-> a memory cannot block an edit. Name the real control, add the"
                  " CODEOWNERS rule, or drop the hazard.%s" % (DIM, OFF))
    if secret_hits or pii_hits or soft_hits:
        print("\n%spossible sensitive data in .memory/:%s" % (BOLD, OFF))
        for where, kind, masked in secret_hits + pii_hits:
            print("  %s  %s  [%s]" % (where, kind, masked))
        for where, kind, masked in soft_hits:
            print("  %s%s  %s  [%s]%s" % (DIM, where, kind, masked, OFF))
        print("  %s-> .memory/ rides the PR and is compiled into AGENTS.md, so anything here is"
              " committed and delivered. Name the secret; keep the value in .env.%s" % (DIM, OFF))
    if contradictions:
        print("\n%stwo live memories contradict each other:%s" % (BOLD, OFF))
        for a, b in contradictions:
            print("  claim: %s" % a.get("key", ""))
            print("    %s%s%s  %s" % (BOLD, a["id"], OFF, a["title"]))
            print("      says: %s" % a["value"])
            print("    %s%s%s  %s" % (BOLD, b["id"], OFF, b["title"]))
            print("      says: %s" % b["value"])
            print("    %s-> agentlore will not pick a winner. If one is now wrong: supersede it.%s" % (DIM, OFF))
            print("    %s-> If one was never true: agentlore retract <id> --reason ...%s" % (DIM, OFF))
            print("    %s-> If both hold in different contexts, say so: --coexists-with + --coexists-why.%s"
                  % (DIM, OFF))
    if chain_problems:
        print("\n%sbroken retirement chains:%s" % (BOLD, OFF))
        for e, msg in chain_problems:
            print("  %s%s%s  %s" % (BOLD, e["id"], OFF, e["title"]))
            print("    %s" % msg)
    if coexisting:
        print("\n%sdeclared coexistence -- both true, on purpose:%s" % (DIM, OFF))
        for a, b in coexisting:
            print("%s  %s vs %s  -- %s%s" % (DIM, a["id"], b["id"],
                  a.get("coexists_why") or b.get("coexists_why") or "?", OFF))
    if unresolved_since:
        print("\n%s--since %s does not resolve in this clone.%s" % (BOLD, unresolved_since, OFF))
        print("    %s-> the dead-ends-settled check is blind without it, so this fails on purpose.%s"
              % (DIM, OFF))
        print("    %s-> fetch the base commit (checkout with fetch-depth: 0), or drop --since.%s"
              % (DIM, OFF))
    if unvalued:
        print("\n%sclaims with no value -- nothing can be compared with them:%s" % (DIM, OFF))
        for e in unvalued:
            print("%s  %s  key: %s%s" % (DIM, e["id"], e.get("key", ""), OFF))
        print("%s-> add --value, or drop the --key.%s" % (DIM, OFF))
    if restyled:
        print("\n%sreformatted (bytes changed, content equivalent -- no action needed):%s"
              % (DIM, OFF))
        for e in restyled:
            print("%s  %s  %s%s" % (DIM, e["id"], e["title"], OFF))
    print("\n%snote: memory is a hint. code and failing tests win.%s" % (DIM, OFF))
    if findings:
        print("%s%d finding(s) -> %s%s" % (DIM, len(findings),
              "annotations on the diff + job summary" if args.github else "fix before merging", OFF))
    if args.github:
        emit_github(findings, passed, total, len(entries), len(stale))
    return 0 if passed == total else 1


# -------------------------------------------------------------------- recall

def cmd_recall(args):
    task = " ".join(args.task)
    terms = [t for t in re.findall(r"[a-z0-9]+", task.lower()) if len(t) > 2]
    rows = load_all()
    scored = []
    for e in rows:
        if e["status"] in DEAD_STATUSES and not args.all:
            continue
        blob_title = (e["title"] + " " + " ".join(e["anchors"])).lower()
        blob_body = e["body"].lower()
        score = sum(4 for t in terms if t in blob_title)
        score += sum(1 for t in terms if t in blob_body)
        if score and e["status"] == "accepted":
            score += 2
        if score:
            scored.append((score, e))
    scored.sort(key=lambda x: -x[0])
    picked = scored[: args.limit]
    if not picked:
        print("REPO MEMORY (%s): nothing relevant to %r" % (repo_name(), task))
        return 0
    print("REPO MEMORY (%s) -- %d of %d memories, ranked for %r" % (repo_name(), len(picked), len(rows), task))
    budget = args.budget
    for i, (score, e) in enumerate(picked):
        body = " ".join(e["body"].split())
        if len(body) > 200:
            body = body[:197] + "..."
        block = "[%s] %s/%s  scope: %s\n%s\n  %s" % (
            e["id"], e["type"], e["status"], ", ".join(e["anchors"]) or "-",
            e["title"], body)
        if e.get("evidence"):
            block += "\n  evidence: " + e["evidence"]
        if len(block) + 2 > budget:
            print("  ... %d more trimmed for context budget" % (len(picked) - i))
            break
        budget -= len(block) + 2
        print("\n" + block)
    return 0


# ------------------------------------------------------------------- compile

def _use_instead(body_text):
    """Pull the actionable half out of a dead-end body: the convention it names.

    Dead-end bodies end with the thing to do instead, because "we tried X" is only useful
    to an agent if it also says what to do. Returns "" when the body names nothing.
    """
    flat = " ".join((body_text or "").split())
    m = re.search(r"Use instead:\s*(.+?)(?:\.\s|$|\.$)", flat)
    if m:
        return "Use instead: %s" % m.group(1).strip().rstrip(".")
    return ""


def compile_block():
    """The always-on memory, as the one block that lands in AGENTS.md.

    Conventions, hazards, and live dead ends. Dead ends were left out at first, which
    put the highest-value entry in the repo -- "we tried this and it was wrong" -- in a
    file nothing pointed the agent at, so an agent could do a full task without ever
    learning the team had already rejected the design it was about to write.

    Kept separate from cmd_compile so the `compile-current` check can ask what compile
    WOULD produce, without writing anything.
    """
    entries = [e for e in load_all() if e["type"] == "convention" and e["status"] == "accepted"]
    hazards = [e for e in load_all() if e["type"] == "hazard" and e["status"] == "accepted"]
    # Live only: a dead end that was settled or retired is history, not a warning, so
    # compiling it would re-warn about a condition that no longer holds.
    dead = [e for e in load_all()
            if e["type"] == "dead-end" and e["status"] == "accepted"
            and not (e.get("resolved_by") or "").strip()]
    body = ["<!-- agentlore:begin (generated from .memory/ - do not edit) -->",
            "## House rules", ""]
    if not entries:
        body.append("_(none recorded)_")
    for e in entries:
        scope = (" `%s`" % ", ".join(e["anchors"])) if e["anchors"] else ""
        body.append("- **%s**%s" % (e["title"], scope))
        if e["body"]:
            body.append("  %s" % " ".join(e["body"].split()))
    # Hazards compile differently on purpose: they are stated as frozen areas with the
    # control named, so the agent knows this is a platform boundary and not a suggestion
    # it may weigh against the task.
    if hazards:
        body += ["", "### Frozen areas", "",
                 "_Do not edit these. The enforcement lives in the repo's controls, not in "
                 "this file._", ""]
        for e in hazards:
            scope = (" `%s`" % ", ".join(e["anchors"])) if e["anchors"] else ""
            who = (" -- owner: %s" % e["owner"]) if e.get("owner") else ""
            enf = ("; enforced by: %s" % e["enforcement"]) if e.get("enforcement") else ""
            body.append("- **%s**%s%s%s" % (e["title"], scope, who, enf))
    if dead:
        body += ["", "### Rejected approaches", "",
                 "_Tried and rejected. Do not re-try these. The reason and the way out are "
                 "recorded in `.memory/dead-ends.md`._", ""]
        for e in dead[:COMPILE_DEAD_MAX]:
            scope = (" `%s`" % ", ".join(e["anchors"])) if e["anchors"] else ""
            body.append("- **%s**%s" % (e["title"], scope))
            use = _use_instead(e["body"])
            if use:
                body.append("  %s" % use)
        if len(dead) > COMPILE_DEAD_MAX:
            body += ["", "_%d more in `.memory/dead-ends.md`._" % (len(dead) - COMPILE_DEAD_MAX)]
    body += ["", "<!-- agentlore:end -->"]
    return "\n".join(body)


def cmd_compile(args):
    """Write the compiled block into AGENTS.md, replacing any previous block in place."""
    agent = Path("AGENTS.md")
    block = compile_block()
    text = agent.read_text() if agent.exists() else "# AGENTS.md\n"
    if "<!-- agentlore:begin" in text:
        text = re.sub(r"<!-- agentlore:begin.*?<!-- agentlore:end -->", block, text, flags=re.S)
    else:
        text = text.rstrip("\n") + "\n\n" + block + "\n"
    agent.write_text(text)
    _all = load_all()
    _conv = len([e for e in _all if e["type"] == "convention" and e["status"] == "accepted"])
    _haz = len([e for e in _all if e["type"] == "hazard" and e["status"] == "accepted"])
    _dead = len([e for e in _all if e["type"] == "dead-end" and e["status"] == "accepted"
                 and not (e.get("resolved_by") or "").strip()])
    print("compiled %d conventions, %d hazards, %d dead ends into AGENTS.md"
          % (_conv, _haz, min(_dead, COMPILE_DEAD_MAX)))
    return 0


# --------------------------------------------------------------- supersede

def cmd_supersede(args):
    """Mark a memory superseded by a new one: the reason memories stop lying."""
    target, candidates = resolve_id(args.old)
    if target is None:
        return refuse_id(args.old, candidates)
    new_id = args.new
    for name in SOURCES:
        p = MEM / name
        if not p.exists():
            continue
        blocks = re.split(r"(?m)(?=^##\s+)", p.read_text())
        hit = False
        rebuilt = []
        for b in blocks:
            # scope the edit to the ONE entry with this id, never the file header
            if b.startswith("## ") and _id_in_block(b, target):
                hit = True
                if "status:" in b:
                    b = re.sub(r"(?m)^status:.*$", "status: superseded", b)
                else:
                    b = re.sub(r"(?m)^(type:.*)$", r"\1\nstatus: superseded", b, count=1)
                if "superseded_by:" not in b:
                    b = re.sub(r"(?m)^(status:.*)$", r"\1\nsuperseded_by: " + new_id, b, count=1)
                # keep two same-titled entries distinguishable in review
                b = re.sub(r"(?m)^##\s+(.+?)\s*$", r"## \1 (superseded)", b, count=1)
            rebuilt.append(b)
        if hit:
            p.write_text("".join(rebuilt))
            print("marked %s superseded by %s in %s" % (target, new_id, name))
            print("  it stays in history as '(superseded)'; reviewers see why it changed.")
            break
    else:
        print("no memory with id %s" % target)
        return 1
    return cmd_index(args)


# --------------------------------------------------------------- retract / rm

def cmd_retract(args):
    """Mark a memory as never having been true.

    Deliberately NOT a delete. The common case is not "this was true and has changed" --
    it is "this should never have been recorded at all" (an agent inferred something
    plausible and wrong). Deleting it destroys the evidence a reviewer needs, so the text
    stays and the entry is labelled, exactly like a supersession.
    """
    if not (args.reason or "").strip():
        print("--reason is required.")
        print("  %sA memory that disappears with no explanation is worse than one that was"
              " wrong:%s" % (DIM, OFF))
        print("  %sthe next agent cannot tell whether it was retracted or lost.%s" % (DIM, OFF))
        return 1
    rid, candidates = resolve_id(args.id)
    if rid is None:
        return refuse_id(args.id, candidates)
    for name in SOURCES:
        p = MEM / name
        if not p.exists():
            continue
        blocks = re.split(r"(?m)(?=^##\s+)", p.read_text())
        rebuilt, hit = [], False
        for i, b in enumerate(blocks):
            # index 0 is the file header, never an entry
            if i > 0 and b.startswith("## ") and _id_in_block(b, rid):
                hit = True
                if "status:" in b:
                    b = re.sub(r"(?m)^status:.*$", "status: retracted", b)
                else:
                    b = re.sub(r"(?m)^(type:.*)$", r"\1\nstatus: retracted", b, count=1)
                if "retracted_reason:" in b:
                    b = re.sub(r"(?m)^retracted_reason:.*$",
                               "retracted_reason: " + args.reason, b)
                else:
                    b = re.sub(r"(?m)^(status:.*)$",
                               r"\1\nretracted_reason: " + args.reason, b, count=1)
                if not re.search(r"(?m)^##\s+.*\((retracted|superseded)\)\s*$", b):
                    b = re.sub(r"(?m)^##\s+(.+?)\s*$", r"## \1 (retracted)", b, count=1)
            rebuilt.append(b)
        if hit:
            p.write_text("".join(rebuilt))
            print("retracted %s in %s" % (args.id, name))
            print("  %skept on disk as '(retracted)' with the reason, and excluded from live"
                  " checks.%s" % (DIM, OFF))
            return cmd_index(args)
    print("no memory with id %s" % args.id)
    return 1


def cmd_rm(args):
    """Delete a memory block outright.

    For one case only: text that must not be in the repository at all (a pasted customer
    name, a token, an internal URL). Everywhere else `retract` is the right verb, because
    deleting a memory removes the evidence along with the mistake.

    The honest limit is printed every time: this rewrites the file, not history. In a
    shared repo the text is already in every clone, and `git log -p` still has it.
    """
    rid, candidates = resolve_id(args.id)
    if rid is None:
        return refuse_id(args.id, candidates)
    for name in SOURCES:
        p = MEM / name
        if not p.exists():
            continue
        blocks = re.split(r"(?m)(?=^##\s+)", p.read_text())
        keep, hit = [], False
        for i, b in enumerate(blocks):
            # only ever drop a real entry block; the file header is not a memory
            if i > 0 and b.startswith("## ") and _id_in_block(b, rid):
                hit = True
                continue
            keep.append(b)
        if hit:
            text = "".join(keep)
            text = re.sub(r"\n{3,}\Z", "\n", text)      # tidy only the tail
            if not text.endswith("\n"):
                text += "\n"
            p.write_text(text)
            print("removed %s from %s" % (args.id, name))
            print("  %s! this rewrites the file, not history: the text is still in every"
                  " existing clone and in `git log -p`.%s" % (RED, OFF))
            print("  %s  if it was sensitive, removing it here does not unpublish it.%s"
                  % (RED, OFF))
            print("  %s-> for 'this was never true', prefer: agentlore retract <id> --reason ...%s"
                  % (DIM, OFF))
            return cmd_index(args)
    print("no memory with id %s" % args.id)
    return 1


# --------------------------------------------------------------------- list

def cmd_verify(args):
    """Re-stamp a memory against current HEAD, optionally narrowing its scope.

    This is the cheap resolution for a STALE flag: without it the health gate
    is just nagging, and nagging gates get disabled.
    """
    if args.coexists_with and not (args.coexists_why or "").strip():
        print("--coexists-with needs --coexists-why.")
        print("  Declaring two contradictory memories coexistent silences a real conflict,")
        print("  so the declaration has to carry its reason. Nothing gets switched off")
        print("  anonymously -- same rule as retract.")
        return 1
    rev = head_rev()
    if not rev:
        print("not a git repo / no commits -- cannot stamp a revision")
        return 1
    entries = load_all()
    if args.all:
        ids = [e["id"] for e in entries if e["status"] not in DEAD_STATUSES]
    else:
        ids = []
        for needle in args.ids:
            rid, candidates = resolve_id(needle, entries)
            if rid is None:
                return refuse_id(needle, candidates)
            ids.append(rid)
    if not ids:
        print("nothing to verify (pass ids, or --all)")
        return 1
    done = []
    for name in SOURCES:
        p = MEM / name
        if not p.exists():
            continue
        text = p.read_text()
        blocks = re.split(r"(?m)(?=^##\s+)", text)
        rebuilt = []
        for b in blocks:
            if b.startswith("## ") and any(_id_in_block(b, eid) for eid in ids):
                if "verified_at:" in b:
                    b = re.sub(r"(?m)^verified_at:.*$", "verified_at: " + rev, b)
                else:
                    b = re.sub(r"(?m)^(type:.*)$", r"\1\nverified_at: " + rev, b, count=1)
                if args.anchor:
                    if "anchors:" in b:
                        b = re.sub(r"(?m)^anchors:.*$", "anchors: " + " ".join(args.anchor), b)
                    else:
                        b = re.sub(r"(?m)^(status:.*)$", r"\1\nanchors: " + " ".join(args.anchor), b, count=1)
                # re-fingerprint the (possibly narrowed) anchor set
                am = re.search(r"(?m)^anchors:\s*(.*)$", b)
                scope = [x for x in re.split(r"[,\s]+", am.group(1)) if x] if am else []
                raw_fp, norm_fp = anchor_fingerprints({"anchors": scope})
                if "anchor_hash:" in b:
                    b = re.sub(r"(?m)^anchor_hash:.*$", "anchor_hash: " + raw_fp, b)
                else:
                    b = re.sub(r"(?m)^(verified_at:.*)$", r"\1\nanchor_hash: " + raw_fp, b, count=1)
                if "anchor_norm:" in b:
                    b = re.sub(r"(?m)^anchor_norm:.*$", "anchor_norm: " + norm_fp, b)
                else:
                    b = re.sub(r"(?m)^(anchor_hash:.*)$", r"\1\nanchor_norm: " + norm_fp, b, count=1)
                if args.resolved_by:
                    if "resolved_by:" in b:
                        b = re.sub(r"(?m)^resolved_by:.*$", "resolved_by: " + args.resolved_by, b)
                    else:
                        b = re.sub(r"(?m)^(anchor_norm:.*)$",
                                   r"\1\nresolved_by: " + args.resolved_by, b, count=1)
                    # A settled dead end must not keep reading as a live warning. Agents read
                    # the markdown directly, so the HEADING has to carry that fact: an honest
                    # body is not enough. Observed in practice -- an agent added resolved_by and
                    # a "condition no longer holds" body, and left the title asserting the dead
                    # condition, which is the first thing the next agent reads.
                    if not re.search(r"(?m)^##\s+.*\((settled|superseded)\)\s*$", b):
                        b = re.sub(r"(?m)^##\s+(.+?)\s*$", r"## \1 (settled)", b, count=1)
                # Declaring coexistence has to be possible on an EXISTING entry: re-adding a
                # memory would create a duplicate rather than settle the pair, so the
                # resolution for a contradiction lives here rather than in `add`.
                if args.coexists_with:
                    cur = re.search(r"(?m)^coexists_with:\s*(.*)$", b)
                    have = [x for x in re.split(r"[,\s]+", cur.group(1)) if x] if cur else []
                    merged = " ".join(sorted(set(have) | set(args.coexists_with)))
                    if cur:
                        b = re.sub(r"(?m)^coexists_with:.*$", "coexists_with: " + merged, b)
                    else:
                        b = re.sub(r"(?m)^(status:.*)$", r"\1\ncoexists_with: " + merged,
                                   b, count=1)
                    if "coexists_why:" in b:
                        b = re.sub(r"(?m)^coexists_why:.*$",
                                   "coexists_why: " + args.coexists_why, b)
                    else:
                        b = re.sub(r"(?m)^(coexists_with:.*)$",
                                   r"\1\ncoexists_why: " + args.coexists_why, b, count=1)
                done.append(re.search(r"(?m)^id:\s*(\S+)", b).group(1))
            rebuilt.append(b)
        p.write_text("".join(rebuilt))
    for eid in done:
        print("verified %s @ %s%s" % (eid, rev, ("  scope -> " + " ".join(args.anchor)) if args.anchor else ""))
    if not done:
        print("no live memory matched")
        return 1
    return cmd_index(args)


def cmd_list(args):
    rows = load_all()
    if not rows:
        print("no memories yet")
        return 0
    print("%-24s %-10s %-11s %-28s %s" % ("ID", "TYPE", "STATUS", "SCOPE", "TITLE"))
    for e in rows:
        print("%-24s %-10s %-11s %-28s %s" % (
            e["id"], e["type"], e["status"],
            (", ".join(e["anchors"]) or "-")[:28], e["title"]))
    print("\n%d memories" % len(rows))
    return 0


# --------------------------------------------------------------------- wiring

def main(argv=None):
    p = argparse.ArgumentParser(prog="agentlore", description=__doc__.splitlines()[0],
                                formatter_class=argparse.RawDescriptionHelpFormatter)
    p.add_argument("--version", action="version", version="agentlore " + VERSION)
    sub = p.add_subparsers(dest="cmd")

    s = sub.add_parser("init", help="create .memory/ and gitignore the index")
    s.set_defaults(func=cmd_init)

    s = sub.add_parser("add", help="record a typed memory (watch the diff)")
    s.add_argument("--type", choices=["decision", "dead-end", "convention", "hazard"], required=True)
    s.add_argument("--title", required=True)
    s.add_argument("--body", default="")
    s.add_argument("--anchor", action="append", default=[], help="file glob this memory is about (repeatable)")
    s.add_argument("--evidence", default="", help="required for dead-end: the commit/PR/error that proved it")
    s.add_argument("--supersedes", action="append", default=[], help="id of a memory this replaces")
    s.add_argument("--key", default="",
                   help="what this memory makes a claim about, e.g. refund.window_days "
                        "(two live memories with the same key must agree on --value)")
    s.add_argument("--value", default="", help="the claim itself, e.g. 90 (pair with --key)")
    s.add_argument("--coexists-with", action="append", default=[],
                   help="id of a live memory that contradicts this one for a real reason "
                        "(context-dependent facts); requires --coexists-why")
    s.add_argument("--coexists-why", default="",
                   help="why both of the contradictory claims are true")
    s.add_argument("--allow-directive", default=None,
                   help="record a directive this memory is allowed to contain, and why "
                        "(required to exempt an entry from the no-agent-directives check)")
    s.add_argument("--resolved-by", default=None,
                   help="dead-end only: what settled it (commit/PR). Omit if it still holds.")
    s.add_argument("--owner", default="",
                   help="hazard only: the team/role accountable for the frozen area")
    s.add_argument("--enforcement", default="",
                   help="hazard only: the control that actually enforces it (e.g. CODEOWNERS)")
    s.add_argument("--status", default="accepted")
    s.add_argument("--author", default="")
    s.add_argument("--review-by", default="")
    s.set_defaults(func=cmd_add)

    s = sub.add_parser("index", help="rebuild the derived SQLite index")
    s.set_defaults(func=cmd_index)

    s = sub.add_parser("check", help="MEMORY-HEALTH report (use in CI, exit 1 when broken)")
    s.add_argument("--brief", action="store_true", help="one line, for session start")
    s.add_argument("--github", action="store_true",
                   help="emit GitHub Actions annotations + job summary (CI mode)")
    s.add_argument("--since", default="",
                   help="the ref this change starts from (CI passes the PR base sha)")
    s.set_defaults(func=cmd_check)

    s = sub.add_parser("recall", help="bounded context block for a task")
    s.add_argument("task", nargs="+")
    s.add_argument("--limit", type=int, default=6)
    s.add_argument("--budget", type=int, default=1600)
    s.add_argument("--all", action="store_true", help="include superseded memories")
    s.set_defaults(func=cmd_recall)

    s = sub.add_parser("compile", help="emit conventions into AGENTS.md")
    s.set_defaults(func=cmd_compile)

    s = sub.add_parser("supersede", help="retire a memory by pointing it at its replacement")
    s.add_argument("old")
    s.add_argument("new")
    s.set_defaults(func=cmd_supersede)

    s = sub.add_parser("retract",
                       help="mark a memory as never having been true (keeps the text)")
    s.add_argument("id")
    s.add_argument("--reason", required=True,
                   help="why this was wrong. Required: a silent disappearance is worse "
                        "than a wrong memory.")
    s.set_defaults(func=cmd_retract)

    s = sub.add_parser("rm",
                       help="delete a memory block outright (text that must not be in the "
                            "repo at all)")
    s.add_argument("id")
    s.set_defaults(func=cmd_rm)

    s = sub.add_parser("verify", help="re-stamp memories against HEAD (clear stale flags)")
    s.add_argument("ids", nargs="*")
    s.add_argument("--all", action="store_true")
    s.add_argument("--anchor", action="append", default=[], help="also narrow/reset the scope")
    s.add_argument("--resolved-by", default=None,
                   help="also record what settled this dead end")
    s.add_argument("--coexists-with", action="append", default=[],
                   help="declare that this memory and another live memory contradict each "
                        "other on purpose (context-dependent facts); requires --coexists-why")
    s.add_argument("--coexists-why", default="", help="why both contradictory claims are true")
    s.set_defaults(func=cmd_verify)

    s = sub.add_parser("list", help="list memories")
    s.set_defaults(func=cmd_list)

    args = p.parse_args(argv)
    if not getattr(args, "func", None):
        p.print_help()
        return 0
    # Fail closed on an empty --resolved-by.
    #
    # The flag exists to record WHY a memory was settled, and its help says to omit it if
    # the memory still holds. An empty value is therefore a mistake -- typically an unset
    # shell variable, as in --resolved-by "$PR_TITLE" -- and treating it as an omission
    # silently re-stamps the staleness clock while the caller believes an audit trail was
    # written. That is the same fail-open class as an ambiguous id or an unresolvable
    # --since, and it is refused the same way.
    _rb = getattr(args, "resolved_by", None)
    if _rb is not None and not _rb.strip():
        print("--resolved-by was given but is empty.")
        print("  it records WHY a memory was settled, so an empty one leaves the change")
        print("  with no audit trail. Omit the flag if the memory still holds, or state")
        print("  what settled it: --resolved-by \"PR #123 removed the fallback\"")
        return 1
    if hasattr(args, "author") and not args.author:
        args.author = _run(["git", "config", "user.name"]) or "unknown"
    try:
        return args.func(args)
    except BrokenPipeError:
        # `agentlore list | head` is a normal thing to do; a traceback is not. Redirect the
        # dead stdout to devnull so Python does not also complain while shutting down.
        os.dup2(os.open(os.devnull, os.O_WRONLY), sys.stdout.fileno())
        return 0


if __name__ == "__main__":
    sys.exit(main())
