"""Roadmap #28 Part 1 — auto-link RFC references in submitted prose. Scans plain-text PR descriptions and comment bodies for references to existing **accepted** (state='active') RFCs and returns a structured list of *segments* the frontend renders: plain-text runs interleaved with ``{"type": "rfc", ...}`` link segments. The backend never emits HTML — the frontend maps link segments onto React anchors — so the surface is XSS-safe by construction and independent of any HTML-sanitization layer. **Read-time enrichment, not submit-time persistence.** The roadmap row phrases the scan as happening "at submit/post time"; this module instead enriches on read. The intent the roadmap actually names — "not as live compose preview" — is honored (drafts are never scanned, only submitted content on the read paths). Read-time was chosen for three reasons: 1. Correctness — links track the *live* active-RFC set. A newly-accepted RFC starts linking in older comments; a withdrawn RFC stops linking everywhere. Submit-time freezing would drift stale. 2. Zero migration — no derived data to store. (A concurrent session already holds migration 023; staying migration-free keeps this slice conflict-free as well as simpler.) 3. Cost — the active-RFC corpus is small and cache-resident, so building the term index and scanning a ≤20k-char body per read is cheap. **Matching is conservative by design.** Only references that are unlikely to be coincidental link: * ``rfc_id`` tokens (e.g. ``RFC-0001``) — inherently specific. * Multi-word titles (containing whitespace, e.g. ``Open Human Model``). * Hyphenated slugs (containing ``-``, e.g. ``open-human-model``). Single common-word titles or slugs (e.g. a hypothetical RFC titled "Human") are deliberately NOT auto-linked — they would turn every prose "human" into a link. Surfacing those is the job of the roadmap's "curated canonical-terms list", an explicit per-deployment opt-in left as a future extension rather than guessed at here. """ from __future__ import annotations from typing import Any, Iterable def _is_word_char(c: str) -> bool: """Word-boundary test. Hyphen and underscore count as word chars so a match can't begin or end in the middle of a kebab/snake token.""" return c.isalnum() or c in ("-", "_") def segment_text(text: str | None, terms: list[tuple[str, str, str]]) -> list[dict[str, Any]]: """Split ``text`` into text / rfc-link segments against ``terms``. ``terms`` is a list of ``(key_lower, slug, title)`` tuples; callers pass it pre-sorted longest-first so the longest match wins at any position (so "Open Human Model" wins over a bare "Open"). Matching is case-insensitive and respects word boundaries on both ends. The returned ``label`` preserves the source casing. Always returns at least one segment; for empty/None input that is a single empty text segment, so callers can render uniformly. """ if not text: return [{"type": "text", "text": text or ""}] out: list[dict[str, Any]] = [] buf: list[str] = [] low = text.lower() n = len(text) i = 0 while i < n: match: tuple[str, str, str, int] | None = None for key, slug, title in terms: klen = len(key) if klen == 0 or not low.startswith(key, i): continue before = text[i - 1] if i > 0 else "" after = text[i + klen] if i + klen < n else "" if _is_word_char(before) or _is_word_char(after): continue match = (key, slug, title, klen) break if match is not None: _key, slug, title, klen = match if buf: out.append({"type": "text", "text": "".join(buf)}) buf = [] out.append({ "type": "rfc", "slug": slug, "label": text[i:i + klen], "title": title, }) i += klen else: buf.append(text[i]) i += 1 if buf: out.append({"type": "text", "text": "".join(buf)}) return out def _keys_for(slug: str, title: str, rfc_id: str | None) -> Iterable[str]: """The match keys an active RFC contributes. See the module docstring for why each gate exists (conservative, false-positive-averse).""" if rfc_id: rid = rfc_id.strip() if len(rid) >= 2: yield rid.lower() if title: t = title.strip() # Multi-word titles only — a single common word is too noisy. if len(t) >= 2 and (" " in t or "\t" in t): yield t.lower() if slug: s = slug.strip() # Hyphenated slugs only — a single-token slug is a bare word. if len(s) >= 2 and "-" in s: yield s.lower() class LinkIndex: """A reusable term index built once per request and applied to many bodies (a PR's description plus every comment on it).""" def __init__(self, terms: list[tuple[str, str, str]]): # Longest key first so the longest reference wins at each position. self._terms = sorted(terms, key=lambda t: len(t[0]), reverse=True) def __bool__(self) -> bool: return bool(self._terms) def segment(self, text: str | None) -> list[dict[str, Any]]: return segment_text(text, self._terms) def build_index(conn, *, exclude_slug: str | None = None) -> LinkIndex: """Build a :class:`LinkIndex` from the accepted (active) RFC corpus. ``exclude_slug`` drops the RFC the surrounding surface is itself scoped to, so an RFC's own title/id/slug don't self-link inside its own PR or discussion. ``ORDER BY slug`` makes key de-duplication deterministic when two RFCs would contribute the same key (first slug wins).""" rows = conn.execute( "SELECT slug, title, rfc_id FROM cached_rfcs WHERE state = 'active' ORDER BY slug" ).fetchall() terms: list[tuple[str, str, str]] = [] seen: set[str] = set() for r in rows: slug = r["slug"] if exclude_slug is not None and slug == exclude_slug: continue title = r["title"] or "" rfc_id = r["rfc_id"] if "rfc_id" in r.keys() else None for key in _keys_for(slug, title, rfc_id): if key in seen: continue seen.add(key) terms.append((key, slug, title)) return LinkIndex(terms)