Files
rfc-app/backend/app/projects.py
T
Ben Stull 4e7410f90b fix(§22/G-15): make the branch/edit/body subsystem three-tier aware (v0.53.0)
§22 migrated the READ/catalog path to the three-tier (project/collection)
model but the WRITE/branch/body subsystem still hardcoded the default
project's default collection — resolving every meta-resident entry to the
default content repo at rfcs/<slug>.md, ignoring the entry's project (its own
content repo) and collection (a <subfolder>/rfcs/ prefix). An entry outside
the default collection rendered a blank canonical body and every edit/PR/
body-write path hit the wrong file.

- New single resolver: projects.content_repo_for_collection() +
  projects.entry_location(config, cid, slug) -> (org, repo, md_path)
  (collection -> project -> content_repo + subfolder; falls back to the
  default repo for a legacy/unknown collection).
- Every entry write/read path resolves via it: api_branches (body GET + all
  branch write paths), api_prs (pr-draft/open/merge/withdraw/review/
  resolution-branch), api_graduation (graduate/claim/retire/unretire +
  orchestrator + state-flip), api.py mark-reviewed + idea-PR merge/decline/
  withdraw + proposal preview, api_metadata. bot.open_metadata_pr and
  bot.mark_entry_reviewed gained a file_path param. refresh_meta_branches,
  the webhook corpus-refresh dispatch, and the hygiene branch-delete now
  span every project's content repo, not just the default.
- New additive collection-scoped body-read routes:
  GET /api/projects/{pid}/collections/{cid}/rfcs/{slug}/main and
  .../branches/{branch} disambiguate a slug across collections (G-5) and read
  the entry's own repo. Slug-only routes kept (now collection-aware via the
  cached row). Frontend getRFCMain/getBranch take optional pid+cid; RFCView
  threads them (mirrors the v0.52.1 entry-detail fix).

No migration, no config change. Existing default-collection entries
unaffected. Tests: backend resolver + collection-scoped branch/body + graduate-
in-subfolder write path; frontend api unit. backend 677 / frontend 60 green.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-08 22:39:46 -07:00

243 lines
11 KiB
Python

"""Project registry — the §22 multi-project layer.
A deployment hosts one or more projects (§22.1). The git registry mirror
that lets a deployment declare projects lands in M3 and drives this module.
`seed_default_project` (the §22.13 META_REPO backfill) is retired in M3;
the registry mirror (`registry.refresh_registry`) is authoritative.
"""
from __future__ import annotations
import logging
from . import db
from .config import Config
log = logging.getLogger(__name__)
DEFAULT_PROJECT_ID = "default"
def resolved_default_id(config: Config) -> str:
"""The id of the deployment's bootstrap/default project. Plan A: always
'default' (the re-stamp to a config slug rides Plan B). The config knob is
read here so Plan B can flip the resolution without touching call sites."""
return config.default_project_id.strip() or DEFAULT_PROJECT_ID
def restamp_default_project(config: Config) -> None:
"""§22.13 step 1 — one-time rename of the M1 bootstrap project id
(DEFAULT_PROJECT_ID = 'default') to the deployment's configured default id
(the DEFAULT_PROJECT_ID env var, e.g. 'ohm'), so the deployment's original
corpus lands at a meaningful `/p/<id>/` and `default` is never a public URL.
Renames `project_id` across every project-scoped table (discovered by
column, so it stays correct as the schema grows), then drops the stale
bootstrap `projects` row (its data has moved to the configured row, which
the registry mirror already created). Idempotent and a no-op when the
configured id is still 'default' or no bootstrap rows remain. Runs at
startup after the registry mirror, with FK enforcement off for the rename
(the composite FKs are kept consistent because parent and child rows are
renamed together) and a foreign_key_check backstop before commit.
"""
target = resolved_default_id(config)
if target == DEFAULT_PROJECT_ID:
return
conn = db.conn()
# §22 three-tier: the entry-corpus tables key on collection_id now; detect a
# lingering bootstrap project by the project-grain `collections.project_id`
# (the PRAGMA scan below still renames every project_id column dynamically).
has_rows = conn.execute(
"SELECT 1 FROM collections WHERE project_id = ? LIMIT 1", (DEFAULT_PROJECT_ID,)
).fetchone()
stale_proj = conn.execute(
"SELECT 1 FROM projects WHERE id = ? LIMIT 1", (DEFAULT_PROJECT_ID,)
).fetchone()
if not has_rows and not stale_proj:
return
if conn.execute("SELECT 1 FROM projects WHERE id = ? LIMIT 1", (target,)).fetchone() is None:
log.warning("restamp: target project %r not in registry yet; skipping", target)
return
tables = [r["name"] for r in conn.execute("SELECT name FROM sqlite_master WHERE type='table'")]
pid_tables = [
t for t in tables
if any(c["name"] == "project_id" for c in conn.execute(f"PRAGMA table_info({t})"))
]
conn.execute("PRAGMA foreign_keys = OFF")
try:
conn.execute("BEGIN")
for t in pid_tables:
conn.execute(
f"UPDATE {t} SET project_id = ? WHERE project_id = ?",
(target, DEFAULT_PROJECT_ID),
)
# The bootstrap row's data has moved to the configured (registry) row.
conn.execute("DELETE FROM projects WHERE id = ?", (DEFAULT_PROJECT_ID,))
violations = conn.execute("PRAGMA foreign_key_check").fetchall()
if violations:
conn.execute("ROLLBACK")
raise RuntimeError(
f"restamp left foreign-key violations: {[tuple(v) for v in violations]}"
)
conn.execute("COMMIT")
except Exception:
try:
conn.execute("ROLLBACK")
except Exception:
pass
raise
finally:
conn.execute("PRAGMA foreign_keys = ON")
log.info("restamp: renamed bootstrap project %r -> %r across %d tables",
DEFAULT_PROJECT_ID, target, len(pid_tables))
def reconcile_default_collection_id(config: Config) -> None:
"""Heal the §22 migration-029 vs registry-mirror divergence for the default
project's collection id on a multi-project deployment.
Migration 029 seeds each project's default collection id as the literal
'default' only when the DB holds a single project at migration time; with
≥2 projects it falls back to the *project id* (avoiding a PK collision —
029 can't read DEFAULT_PROJECT_ID, there is no env in SQL). But the registry
mirror (`registry._default_collection_id`) expects the deployment's default
project to own the collection id 'default'. On an upgrade whose DB already
held ≥2 projects when 029 ran, the default project's collection is therefore
named after the project (e.g. 'ohm'), and the next mirror would INSERT a
second, empty 'default' collection instead of merging — duplicating the
default corpus and orphaning the entries (which point at 'ohm').
This is the collection-grain twin of `restamp_default_project`. Run at
startup BEFORE the registry mirror so the canonical 'default' collection
already exists when the mirror upserts (merge, not duplicate). Renames the
divergent collection's id to 'default' and cascades `collection_id` across
every collection-keyed table, with FK enforcement off for the atomic rename
and a `foreign_key_check` backstop before commit. Idempotent; a no-op on
fresh / single-project / already-aligned deployments.
"""
from .collections import DEFAULT_COLLECTION_ID
target = resolved_default_id(config)
if target == DEFAULT_COLLECTION_ID: # default project already owns 'default'
return
conn = db.conn()
# The 029 ≥2-projects seed names the default project's collection after the
# project itself; the canonical id the mirror expects is 'default'.
divergent = conn.execute(
"SELECT 1 FROM collections WHERE id = ? AND project_id = ? LIMIT 1",
(target, target),
).fetchone()
if not divergent:
return
if conn.execute(
"SELECT 1 FROM collections WHERE id = ? LIMIT 1", (DEFAULT_COLLECTION_ID,)
).fetchone():
# A 'default' collection already exists (e.g. a prior buggy mirror left a
# duplicate). Don't auto-merge data — that needs care; leave both for
# operator cleanup and log loudly.
log.warning(
"reconcile: default project %r owns both a %r and a 'default' "
"collection; skipping auto-rename (manual merge required)",
target, target,
)
return
cid_tables = [
t["name"]
for t in conn.execute("SELECT name FROM sqlite_master WHERE type='table'")
if any(c["name"] == "collection_id"
for c in conn.execute(f"PRAGMA table_info({t['name']})"))
]
conn.execute("PRAGMA foreign_keys = OFF")
try:
conn.execute("BEGIN")
conn.execute(
"UPDATE collections SET id = ? WHERE id = ?",
(DEFAULT_COLLECTION_ID, target),
)
for t in cid_tables:
conn.execute(
f"UPDATE {t} SET collection_id = ? WHERE collection_id = ?",
(DEFAULT_COLLECTION_ID, target),
)
violations = conn.execute("PRAGMA foreign_key_check").fetchall()
if violations:
conn.execute("ROLLBACK")
raise RuntimeError(
f"reconcile left foreign-key violations: {[tuple(v) for v in violations]}"
)
conn.execute("COMMIT")
except Exception:
try:
conn.execute("ROLLBACK")
except Exception:
pass
raise
finally:
conn.execute("PRAGMA foreign_keys = ON")
log.info(
"reconcile: renamed default-project collection %r -> 'default' across %d tables",
target, len(cid_tables),
)
def default_content_repo(config: Config) -> str | None:
"""The content repo the single-corpus mirror reads, from the default
project's row (filled by the registry mirror). Replaces the retired
META_REPO. None until the registry mirror has run."""
row = db.conn().execute(
"SELECT content_repo FROM projects WHERE id = ?",
(resolved_default_id(config),),
).fetchone()
return row["content_repo"] if row and row["content_repo"] else None
def content_repo(project_id: str) -> str | None:
"""The content repo for a specific project (§22.3). None if unknown/unset.
The per-project successor to `default_content_repo` for the write path."""
row = db.conn().execute(
"SELECT content_repo FROM projects WHERE id = ?", (project_id,)
).fetchone()
return row["content_repo"] if row and row["content_repo"] else None
def content_repo_for_collection(collection_id: str) -> str | None:
"""The content repo a collection's entries live in (§22 three-tier write
path, G-15): collection → project → content_repo. None if the collection or
its project is unknown/unset. The per-collection successor to
`default_content_repo` for the WRITE path — an entry in a non-default
project must read/write that project's repo, not the deployment default."""
from . import collections as collections_mod
pid = collections_mod.project_of_collection(collection_id)
return content_repo(pid) if pid else None
def entry_location(config: Config, collection_id: str, slug: str) -> tuple[str, str, str]:
"""The git location `(gitea_org, content_repo, md_path)` of an entry, resolved
from its collection (§22 three-tier, G-15).
Repo: the collection's project content_repo, falling back to the deployment
default project's repo when the collection (or its project) is unknown — so a
legacy/single-corpus entry still resolves to a usable location rather than an
empty repo. Path: `<subfolder>/rfcs/<slug>.md`, or `rfcs/<slug>.md` at the
repo root for a default (subfolder-less) collection.
This is the single resolver the branch/edit/body/metadata/graduation write
paths share, replacing the hardcoded `default_content_repo` + `rfcs/<slug>.md`.
"""
from . import collections as collections_mod
repo = content_repo_for_collection(collection_id) or (default_content_repo(config) or "")
sub = collections_mod.subfolder_of(collection_id)
rfcs_dir = f"{sub}/rfcs" if sub else "rfcs"
return config.gitea_org, repo, f"{rfcs_dir}/{slug}.md"
def project_initial_state(project_id: str) -> str:
"""§22.4b landing state for new entries in a project's default collection
(the per-corpus field moved down to the collection in migration 029).
Defaults to 'super-draft' for an unknown/unset row (today's-flow default)."""
from . import collections as collections_mod
return collections_mod.collection_initial_state(
collections_mod.default_collection_id(project_id)
)