Files
human-experience-filter-art/tools/pipeline/track.py
T
BenStullsBets 70fd367c70 tools(pipeline): hybrid track.py + simulator label author mode
Content-pipeline Increment 2, part 2 (the tooling). Implements design §11.5
(docs/superpowers/specs/2026-06-24-content-pipeline-design.md):

- tools/pipeline/track.py — the stage-5 geometry pass. Classical path: a
  hand-seeded normalized box propagated by OpenCV Lucas-Kanade optical flow
  (CSRT used instead when a contrib build provides it), sampled to a SPARSE
  loop-normalized keyframed track + an appear/disappear window. Optional ML
  detect+track path lazy-imports ultralytics (clear ImportError if absent;
  base install needs only cv2). Pure helpers (normalize/denormalize, loop_t,
  infer_window, sample_track, track_to_annotation) are unit-tested; the cv2
  propagation is an opt-in integration test on real abyss_wow footage.
  Semantics are never produced here — geometry only.

- Author mode — /author.html + author.js/.css reuse the preview stage: pick a
  pool clip, scrub, drag a seed box, Run tracker (or Add as static box), author
  the LEFT detail tiers (general -> scientific+fact) + salience, shift-click to
  place affect anchors with RIGHT emotion tiers, and Save to manifest. Backend:
  POST /api/author/track (runs track_seed on the clip's base) + POST
  /api/author/clip (idempotent upsert via tools.pipeline.manifest — keeps media
  + provenance, replaces only authored content, reloads in place). The tracker
  propagates box geometry; all strings/scientific names/facts are hand-typed.

- Repointed the pipeline integration test off the retired forest base to the
  cosmos pool primary. USER_GUIDE simulator section brought current (pools,
  coast, 3-knob Mood, real-time dream, progressive tiers, author mode).

267 passed / 2 skipped (+ track pure + opt-in real-footage + author endpoint
tests). Author UI by-eye review deferred to the operator (no Chrome on this box).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-06-24 18:14:32 -07:00

298 lines
12 KiB
Python

"""Stage 5 geometry — the hybrid motion-track pass (content-pipeline §3.5 / §11.5).
Turns a hand-seeded box (or an ML detection) on one keyframe into the sparse,
loop-normalized keyframed `track` the manifest stores (design §4) plus an
appear/disappear window (§11.2). SEMANTICS are never produced here — the author
owns the `key`, strings, `salience` and tiers; this module only propagates BOX
GEOMETRY.
Two geometry paths, same artifact:
- **classical** (always available): a hand-seeded box propagated by OpenCV
Lucas-Kanade optical flow (a CSRT tracker is used instead when a contrib
build provides one). Translation model — right for calm drifting subjects
(the abyss/reef creatures). Re-seeds features when too few survive.
- **ML detect+track** (optional, lazy-imported): a YOLO+tracker pass proposing
boxes the author maps to a key. Imported only on demand so the base install
needs no torch/ultralytics; a clear ImportError tells you what to `pip
install` if you ask for it without it present.
The PURE helpers (normalization, keyframe sampling, window inference, annotation
assembly) are unit-tested; the cv2/ML propagation is covered by the opt-in
integration test on real footage.
"""
from __future__ import annotations
from typing import Optional
Box = tuple[float, float, float, float] # (x, y, w, h)
# --- pure geometry helpers --------------------------------------------------
def clamp01_box(box: Box) -> list[float]:
"""Clamp a normalized box into the visible [0,1] frame, keeping it on-screen
(origin clamped, then size trimmed so x+w<=1, y+h<=1)."""
x, y, w, h = box
x = min(max(x, 0.0), 1.0)
y = min(max(y, 0.0), 1.0)
w = min(max(w, 0.0), 1.0 - x)
h = min(max(h, 0.0), 1.0 - y)
return [x, y, w, h]
def normalize_box(box_px: Box, width: int, height: int) -> list[float]:
"""Pixel box -> normalized [0,1] box (the manifest's resolution-independent
form). Clamped on-screen."""
x, y, w, h = box_px
return clamp01_box((x / width, y / height, w / width, h / height))
def denormalize_box(box_norm: Box, width: int, height: int) -> tuple[int, int, int, int]:
"""Normalized [0,1] box -> integer pixel box for the cv2 tracker."""
x, y, w, h = box_norm
return (round(x * width), round(y * height), round(w * width), round(h * height))
def loop_t(frame_idx: int, total_frames: int) -> float:
"""Loop-normalized time of a frame in [0,1), matching the client's
`currentTime / duration` clock (design §4)."""
if total_frames <= 0:
return 0.0
return frame_idx / total_frames
def infer_window(tracked_frames: list[int], total_frames: int) -> tuple[float, float]:
"""The appear/disappear window (§11.2) covering the frames where a box was
tracked, as loop-normalized t. The first/last tracked frame bound it."""
if not tracked_frames:
return (0.0, 1.0)
lo, hi = min(tracked_frames), max(tracked_frames)
return (loop_t(lo, total_frames), loop_t(hi, total_frames))
def sample_track(
per_frame: dict[int, Box], total_frames: int, n_keyframes: int = 5
) -> list[dict]:
"""Down-sample a dense per-frame box map to a SPARSE keyframed track
(design §4: sparse keyframes interpolated at runtime, never dense per-frame).
Picks `n_keyframes` frames evenly across the tracked span (always including
the first and last tracked frame) and emits `[{t, box}, ...]` sorted by t,
de-duplicated on t. Pure — takes already-normalized boxes."""
if not per_frame:
return []
frames = sorted(per_frame)
n = max(2, n_keyframes)
if len(frames) <= n:
picks = frames
else:
lo, hi = frames[0], frames[-1]
step = (hi - lo) / (n - 1)
targets = [lo + round(step * i) for i in range(n)]
# snap each target to the nearest actually-tracked frame
picks = sorted({min(frames, key=lambda f: abs(f - t)) for t in targets})
out, seen = [], set()
for f in picks:
t = round(loop_t(f, total_frames), 4)
if t in seen:
continue
seen.add(t)
out.append({"t": t, "box": clamp01_box(per_frame[f])})
return out
def track_to_annotation(
key: str,
track: list[dict],
*,
salience: int = 4,
appear: Optional[float] = None,
disappear: Optional[float] = None,
) -> dict:
"""Assemble a manifest annotation from a propagated track + the author's
semantics. Geometry (track + window) comes from this module; `key`,
`salience` and the tiered strings are the author's (added separately)."""
ann: dict = {"key": key, "salience": salience, "track": track}
if appear is not None:
ann["appear"] = round(appear, 4)
if disappear is not None:
ann["disappear"] = round(disappear, 4)
return ann
# --- classical propagation (cv2; covered by the opt-in integration test) -----
def _new_csrt():
"""A CSRT tracker if this OpenCV build ships one (contrib), else None — then
the optical-flow path is used. Kept tiny so the import stays lazy."""
import cv2
for factory in ("TrackerCSRT_create",):
if hasattr(cv2, factory):
return getattr(cv2, factory)()
legacy = getattr(cv2, "legacy", None)
if legacy is not None and hasattr(legacy, "TrackerCSRT_create"):
return legacy.TrackerCSRT_create()
return None
def _features_in_box(gray, box_px, max_corners: int = 60):
"""Good-feature points inside a pixel box, in full-frame coords (for LK)."""
import cv2
import numpy as np
x, y, w, h = (int(v) for v in box_px)
h_img, w_img = gray.shape[:2]
x0, y0 = max(x, 0), max(y, 0)
x1, y1 = min(x + w, w_img), min(y + h, h_img)
if x1 - x0 < 4 or y1 - y0 < 4:
return None
roi = gray[y0:y1, x0:x1]
pts = cv2.goodFeaturesToTrack(roi, max_corners, 0.01, 5)
if pts is None:
return None
pts = pts.reshape(-1, 2) + np.array([x0, y0], dtype="float32")
return pts.reshape(-1, 1, 2)
def track_box(
video_path,
seed_box_norm: Box,
*,
seed_frame: int = 0,
max_frames: Optional[int] = None,
min_features: int = 8,
) -> tuple[dict[int, list[float]], int]:
"""Propagate a hand-seeded normalized box FORWARD from `seed_frame` with
optical flow (or CSRT when available). Returns `(per_frame_norm_boxes,
total_frames)` — feed `per_frame` to `sample_track`. Translation model: the
box follows the median displacement of features inside it, re-seeding when too
few survive. Impure (reads the video) — opt-in integration test."""
import cv2
import numpy as np
cap = cv2.VideoCapture(str(video_path))
total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT) or 0)
if seed_frame:
cap.set(cv2.CAP_PROP_POS_FRAMES, seed_frame)
ok, frame = cap.read()
if not ok:
cap.release()
raise ValueError(f"could not read frame {seed_frame} of {video_path}")
h_img, w_img = frame.shape[:2]
box = list(denormalize_box(seed_box_norm, w_img, h_img)) # px [x,y,w,h]
csrt = _new_csrt()
if csrt is not None:
csrt.init(frame, tuple(int(v) for v in box))
prev_gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
pts = None if csrt is not None else _features_in_box(prev_gray, box)
out: dict[int, list[float]] = {seed_frame: normalize_box(box, w_img, h_img)}
fidx = seed_frame
while True:
ok, frame = cap.read()
if not ok:
break
fidx += 1
if csrt is not None:
ok2, b = csrt.update(frame)
if ok2:
box = [b[0], b[1], b[2], b[3]]
else:
gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)
if pts is None or len(pts) < min_features:
pts = _features_in_box(prev_gray, box)
if pts is not None and len(pts):
nxt, status, _ = cv2.calcOpticalFlowPyrLK(prev_gray, gray, pts, None)
if nxt is not None:
good_old = pts[status == 1]
good_new = nxt[status == 1]
if len(good_new) >= 3:
dx = float(np.median(good_new[:, 0] - good_old[:, 0]))
dy = float(np.median(good_new[:, 1] - good_old[:, 1]))
box[0] += dx
box[1] += dy
pts = good_new.reshape(-1, 1, 2)
else:
pts = None
prev_gray = gray
out[fidx] = normalize_box(box, w_img, h_img)
if max_frames is not None and fidx - seed_frame >= max_frames:
break
cap.release()
return out, total
def track_seed(
video_path,
key: str,
seed_box_norm: Box,
*,
seed_frame: int = 0,
salience: int = 4,
n_keyframes: int = 5,
max_frames: Optional[int] = None,
) -> dict:
"""End-to-end classical pass: seed -> propagate -> sample -> annotation
(geometry only; the author adds strings/tiers). Impure (reads the video)."""
per_frame, total = track_box(
video_path, seed_box_norm, seed_frame=seed_frame, max_frames=max_frames
)
track = sample_track(per_frame, total, n_keyframes=n_keyframes)
appear, disappear = infer_window(list(per_frame), total)
return track_to_annotation(
key, track, salience=salience, appear=appear, disappear=disappear
)
# --- optional ML detect+track (lazy-imported) --------------------------------
def detect_and_track(video_path, *, model: str = "yolov8n.pt", classes=None) -> list[dict]:
"""Optional ML path (content-pipeline §3.5): run a detect+track model and
return candidate tracks `[{track_id, label, salience, track:[{t,box}],
appear, disappear}]` for the author to map to keys. Heavy deps are
lazy-imported so the base install stays light.
Raises a clear ImportError naming the install if `ultralytics` is absent —
this path is opt-in by design; the classical `track_seed` needs only cv2."""
try:
from ultralytics import YOLO # noqa: F401
except ImportError as e: # pragma: no cover - exercised only without the dep
raise ImportError(
"the ML detect+track path needs ultralytics — "
"`pip install ultralytics` (optional; the classical track_seed path "
"uses only opencv-python)."
) from e
import cv2 # noqa: F401 (pragma: no cover below — needs the real model)
yolo = YOLO(model) # pragma: no cover
cap = cv2.VideoCapture(str(video_path)) # pragma: no cover
total = int(cap.get(cv2.CAP_PROP_FRAME_COUNT) or 0) # pragma: no cover
cap.release() # pragma: no cover
per_track: dict[int, dict[int, list[float]]] = {} # pragma: no cover
labels: dict[int, str] = {} # pragma: no cover
for fidx, res in enumerate( # pragma: no cover
yolo.track(source=str(video_path), persist=True, classes=classes, stream=True)
):
if res.boxes is None or res.boxes.id is None: # pragma: no cover
continue
w_img, h_img = res.orig_shape[1], res.orig_shape[0] # pragma: no cover
for box, tid, cls in zip( # pragma: no cover
res.boxes.xywh.tolist(), res.boxes.id.tolist(), res.boxes.cls.tolist()
):
cx, cy, bw, bh = box # pragma: no cover
nb = normalize_box((cx - bw / 2, cy - bh / 2, bw, bh), w_img, h_img)
per_track.setdefault(int(tid), {})[fidx] = nb # pragma: no cover
labels[int(tid)] = yolo.names.get(int(cls), str(cls))
out = [] # pragma: no cover
for tid, frames in per_track.items(): # pragma: no cover
track = sample_track(frames, total) # pragma: no cover
appear, disappear = infer_window(list(frames), total) # pragma: no cover
out.append({ # pragma: no cover
"track_id": tid, "label": labels.get(tid, ""), "salience": 4,
"track": track, "appear": round(appear, 4), "disappear": round(disappear, 4),
})
return out # pragma: no cover