You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [4e327784] deploy live constitution dashboard Expose auditable progress and event streaming, configure the production roots and runtime, and make tested main-branch commits the deployment authority. Co-authored-by: Cursor Side A — unified diff (full patch): diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000000000000000000000000000000000000..c9d63a722beba0a0297fc853089332c460ab78dd --- /dev/null +++ b/.dockerignore @@ -0,0 +1,7 @@ +.git +.venv +.hypothesis +__pycache__ +tests +*.json +*.bsp diff --git a/.github/workflows/deploy.yml b/.github/workflows/deploy.yml new file mode 100644 index 0000000000000000000000000000000000000000..76dcdf82d53177c1e47d86b23a54523239d232a6 --- /dev/null +++ b/.github/workflows/deploy.yml @@ -0,0 +1,49 @@ +name: Test and deploy + +on: + push: + branches: [main] + +concurrency: + group: production + cancel-in-progress: false + +permissions: + contents: read + +jobs: + test: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: astral-sh/setup-uv@v6 + with: + enable-cache: true + + - name: Run Python tests + run: uv run pytest -q + + - name: Install Babashka + run: | + curl -fsSL https://raw.githubusercontent.com/babashka/babashka/master/install \ + | sudo bash -s -- --dir /usr/local/bin + + - name: Run process integration tests + run: bb TEST.sh + + deploy: + needs: test + runs-on: ubuntu-latest + environment: + name: production + url: https://token.slug.social + steps: + - uses: actions/checkout@v4 + + - uses: superfly/flyctl-actions/setup-flyctl@master + + - name: Deploy to Fly + run: flyctl deploy --remote-only + env: + FLY_API_TOKEN: ${{ secrets.FLY_API_TOKEN }} diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000000000000000000000000000000000000..c9a5c00782371c19ad5ab5c58cf6f5a8ffec0141 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,17 @@ +FROM ghcr.io/astral-sh/uv:python3.11-bookworm-slim + +RUN apt-get update \ + && apt-get install -y --no-install-recommends git ca-certificates \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /app +COPY pyproject.toml uv.lock ./ +RUN uv sync --frozen --no-install-project + +COPY constitution.py ./ + +ENV PATH="/app/.venv/bin:${PATH}" \ + PYTHONUNBUFFERED="1" + +EXPOSE 8080 +CMD ["python", "constitution.py"] diff --git a/constitution.py b/constitution.py index 4bee9f83663ab7fb36db95129b92b64b4ef57258..a58257e1881b21d1d6fa8e68a3faa222e4f661ef 100644 --- a/constitution.py +++ b/constitution.py @@ -24,12 +24,12 @@ A daily GitHub Action backs up the JSONL ledger to the same repo. Run: uv run constitution.py """ -from decimal import Decimal, getcontext +from decimal import Decimal, getcontext, DefaultContext from datetime import datetime, timezone from fastapi import FastAPI, Request, Response from fastapi.responses import PlainTextResponse, HTMLResponse from starlette.middleware.sessions import SessionMiddleware -import json, time, os, asyncio, httpx, pathlib, subprocess, hashlib, re, fcntl +import json, time, os, asyncio, httpx, pathlib, subprocess, hashlib, re, fcntl, base64 import sympy as sp # type: ignore[reportMissingImports] from tenacity import retry, retry_if_exception, stop_after_attempt, wait_exponential from evaleval import ( @@ -37,6 +37,7 @@ from evaleval import ( exec_event, One, Two, Three, Selector, MORPH, PREPEND, ) +DefaultContext.prec = 50 getcontext().prec = 50 app = FastAPI() @@ -129,14 +130,47 @@ OPENROUTER_BASE_URL = os.environ.get("OPENROUTER_BASE_URL", "https://openrouter. # using the exact same source; their normalized values are committed to every # discovery event. DEFAULT_REPOSITORIES = [ + { + "id": "constitution", + "url": "https://github.com/sortersocial/constitution.git", + "refs": ["refs/heads/**"], + }, { "id": "slug", - "url": "https://github.com/tommy-mor/slug.git", + "url": "https://github.com/sortersocial/slug.git", + "refs": ["refs/heads/**"], + }, + { + "id": "sorter", + "url": "https://github.com/sorterisntonline/sorter.git", + "refs": ["refs/heads/**"], + }, + { + "id": "sorter2", + "url": "https://github.com/sortersocial/sorter2.git", + "refs": ["refs/heads/**"], + }, + { + "id": "sorter-oldest", + "url": "https://github.com/tommy-mor/sorter.git", "refs": ["refs/heads/**"], }, ] DEFAULT_CONTRIBUTORS = { "tommy-mor": ["thmorriss@gmail.com"], + "christopher-whitman": [ + "chris@cwwhitman.com", + "7566903+cwwhitman@users.noreply.github.com", + ], + "jake-chvatal": [ + "jake+github@uln.industries", + "jakechvatal@gmail.com", + "jake@isnt.online", + ], + "lara": ["me@lara.lv"], + "nat-reid": ["nathanielreid@gmail.com"], + "zod": ["jason.p.mcel@gmail.com", "me@zod.tf"], + "jovan": ["jovan@slug.social", "jovan@getcivicai.com"], } REPOSITORIES = json.loads( @@ -147,6 +181,7 @@ CONTRIBUTORS = json.loads( ) GIT_MIRROR_DIR = pathlib.Path(os.environ.get("GIT_MIRROR_DIR", "/data/git")) GIT_TIMEOUT_SECONDS = int(os.environ.get("GIT_TIMEOUT_SECONDS", "120")) +GITHUB_TOKEN = os.environ.get("GITHUB_TOKEN", "") # Council model IDs: slug.social garden rank under this parent (bodies = OpenRouter URLs), then top-up from OpenRouter list. SLUG_SOCIAL_BASE_URL = os.environ.get("SLUG_SOCIAL_BASE_URL", "https://slug.social").rstrip("/") @@ -661,20 +696,30 @@ def _git(repo: pathlib.Path | None, *args: str, input_bytes: bytes | None = None if repo is not None: command += ["-C", str(repo)] command += list(args) + git_env = { + **os.environ, + "GIT_CONFIG_NOSYSTEM": "1", + "GIT_CONFIG_GLOBAL": os.devnull, + "GIT_NO_REPLACE_OBJECTS": "1", + "LC_ALL": "C", + "TZ": "UTC", + } + if GITHUB_TOKEN: + credential = base64.b64encode( + f"x-access-token:{GITHUB_TOKEN}".encode() + ).decode() + git_env.update({ + "GIT_CONFIG_COUNT": "1", + "GIT_CONFIG_KEY_0": "http.https://github.com/.extraHeader", + "GIT_CONFIG_VALUE_0": f"Authorization: Basic {credential}", + }) try: result = subprocess.run( command, input=input_bytes, stdout=subprocess.PIPE, stderr=subprocess.PIPE, - env={ - **os.environ, - "GIT_CONFIG_NOSYSTEM": "1", - "GIT_CONFIG_GLOBAL": os.devnull, - "GIT_NO_REPLACE_OBJECTS": "1", - "LC_ALL": "C", - "TZ": "UTC", - }, + env=git_env, timeout=GIT_TIMEOUT_SECONDS, check=False, ) @@ -844,9 +889,15 @@ def _build_discovery(epoch_n: int, boundary_ms: int, events: list) -> GitDiscove canonical_location = min( locations[qualified_oid], key=lambda x: (x[0], x[1]) ) + # One commit may be reachable from dozens of refs in the same mirror. + # Verify its object once per repository, not once per source ref. + object_locations = { + (str(m), raw_oid): (m, raw_oid) + for _, _, m, raw_oid in locations[qualified_oid] + } object_hashes = { hashlib.sha256(_git(m, "cat-file", "commit", raw_oid)).hexdigest() - for _, _, m, raw_oid in locations[qualified_oid] + for m, raw_oid in object_locations.values() } if len(object_hashes) != 1: raise RuntimeError(f"conflicting Git objects share OID {qualified_oid}") @@ -994,11 +1045,55 @@ async def discover_repositories(epoch_n: int, boundary_ms: int) -> GitDiscovery: SSE_CLIENTS = [] +AUDIT_HISTORY = [] +AUDIT_SEQUENCE = 0 +PROCESS_STATE = { + "running": False, + "phase": "idle", + "progress": 100, + "message": "Waiting for the next epoch", +} + + +def _sse_event(event_name: str, payload: dict) -> str: + return ( + f"event: {event_name}\n" + f"data: {json.dumps(payload, separators=(',', ':'))}\n\n" + ) + + +async def broadcast_audit( + kind: str, + message: str, + *, + progress: int | None = None, + phase: str | None = None, +) -> dict: + global AUDIT_SEQUENCE + AUDIT_SEQUENCE += 1 + if progress is not None: + PROCESS_STATE["progress"] = max(0, min(100, int(progress))) + if phase is not None: + PROCESS_STATE["phase"] = phase + PROCESS_STATE["message"] = message + payload = { + "id": AUDIT_SEQUENCE, + "timestamp_ms": int(time.time() * 1000), + "kind": kind, + "message": message, + **PROCESS_STATE, + } + AUDIT_HISTORY.append(payload) + del AUDIT_HISTORY[:-200] + wire = _sse_event("audit", payload) + for queue in list(SSE_CLIENTS): + await queue.put(wire) + return payload async def broadcast_js(js: str): """Send a JS snippet to all connected SSE clients.""" - for queue in SSE_CLIENTS: + for queue in list(SSE_CLIENTS): await queue.put(js) @@ -1006,10 +1101,29 @@ async def rank_commits(commits: list[dict]): if not commits: return {}, [] - models = await fetch_top_models(n=3) contributors = sorted(set(c["contributor"] for c in commits)) - if len(contributors) > 1 and not models: + if len(contributors) == 1: + await broadcast_audit( + "ranking", + f"Only {contributors[0]} is eligible; rank is 1.0", + progress=90, + phase="finalizing", + ) + return {contributors[0]: Decimal("1")}, [] + if not (OPENROUTER_API_KEY or "").strip(): + raise RuntimeError( + "OPENROUTER_API_KEY is required when multiple contributors need ranking" + ) + + models = await fetch_top_models(n=3) + if not models: raise RuntimeError("no council models available for contributor ranking") + await broadcast_audit( + "council", + f"Council selected: {', '.join(models)}", + progress=35, + phase="ranking", + ) await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ ["div.log-council", f"Council: {', '.join(models)} — {len(commits)} commits"] ])) @@ -1035,6 +1149,11 @@ async def rank_commits(commits: list[dict]): async def compare_fn(i, j): a1, a2 = authors[i], authors[j] + await broadcast_audit( + "comparison", + f"Comparing {a1} with {a2}", + phase="ranking", + ) await broadcast_js(exec_event(Three[Selector("#emission-status")][MORPH][ ["div#emission-status", f"Comparing {a1} vs {a2}…"] ])) @@ -1050,6 +1169,11 @@ async def rank_commits(commits: list[dict]): if winner_weight <= 0 or loser_weight <= 0: raise ValueError("ratio weights must be positive") results.append((w, l, winner_weight, loser_weight)) + await broadcast_audit( + "vote", + f"{model}: {authors[w]} over {authors[l]} ({result['ratio']})", + phase="ranking", + ) await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ ["div.log-vote", ["span.model", model], " — ", @@ -1059,6 +1183,11 @@ async def rank_commits(commits: list[dict]): ] ])) except Exception as e: + await broadcast_audit( + "error", + f"{model} failed: {e}", + phase="error", + ) await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ ["div.log-error", f"⚠ {model}: {e}"] ])) @@ -1068,8 +1197,13 @@ async def rank_commits(commits: list[dict]): async def progress_fn(ev): if ev["phase"] == "spanning_tree": label = f"Spanning tree: {ev['step']}/{ev['total']}" + percent = 35 + round(35 * ev["step"] / max(ev["total"], 1)) else: label = f"Zip pass {ev['pass']}: {ev['step']}/{ev['total']}" + percent = 70 + round(20 * ev["step"] / max(ev["total"], 1)) + await broadcast_audit( + "progress", label, progress=percent, phase="ranking" + ) await broadcast_js(exec_event(Three[Selector("#emission-status")][MORPH][ ["div#emission-status", label] ])) @@ -1083,6 +1217,12 @@ async def rank_commits(commits: list[dict]): scores = rank_centrality(pairs) ranking = {authors[i]: Decimal(str(scores[i])) for i in range(len(authors))} ranking_rows = sorted(ranking.items(), key=lambda x: x[1], reverse=True) + await broadcast_audit( + "ranking", + "Ranking: " + ", ".join(f"{a} {s:.4f}" for a, s in ranking_rows), + progress=90, + phase="finalizing", + ) await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ ["div.log-ranking", ["b", "Ranking: "], @@ -1104,11 +1244,33 @@ def pool_remaining(events: list) -> Decimal: async def run_emission(epoch_n, boundary_ms): + PROCESS_STATE["running"] = True + await broadcast_audit( + "start", + f"Epoch {epoch_n} emission started", + progress=2, + phase="starting", + ) await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ ["div.log-start", f"⚡ Epoch {epoch_n} emission started"] ])) + await broadcast_audit( + "discovery", + "Fetching configured repositories and snapshotting refs", + progress=8, + phase="discovery", + ) discovery = await discover_repositories(epoch_n, boundary_ms) + await broadcast_audit( + "discovery", + ( + f"Discovered {len(discovery.observations)} new commits; " + f"{len(discovery.commits)} are eligible" + ), + progress=30, + phase="discovery", + ) ranking, models = await rank_commits(discovery.commits) def make_emission(events): @@ -1146,6 +1308,13 @@ async def run_emission(epoch_n, boundary_ms): entry = await store.atomic(make_emission) if entry: + PROCESS_STATE["running"] = False + await broadcast_audit( + "complete", + f"Epoch {entry.epoch} complete; emitted {entry.total_emitted} SLG", + progress=100, + phase="idle", + ) await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ ["div.log-amount", f"Pool {entry.pool_before} → emit {entry.total_emitted} → {entry.pool_after}"] @@ -1189,13 +1358,24 @@ async def distribute_usdc(holdings, treasury_balance): async def epoch_loop(): while True: epoch_n, current_start, next_boundary = current_epoch() + processed = {e.epoch for e in store.read() if isinstance(e, Emission)} + if epoch_n >= 0 and epoch_n not in processed: + try: + await run_emission(epoch_n, current_start) + except Exception as exc: + PROCESS_STATE["running"] = False + await broadcast_audit( + "error", + f"Epoch {epoch_n} failed: {exc}; retrying in 60 seconds", + phase="error", + ) + print(f"epoch {epoch_n} emission failed: {exc}", flush=True) + await asyncio.sleep(60) + continue + now = int(time.time() * 1000) wait_ms = next_boundary - now - if wait_ms <= 0: - processed = {e.epoch for e in store.read() if isinstance(e, Emission)} - if epoch_n not in processed and epoch_n >= 0: - await run_emission(epoch_n, current_start) await asyncio.sleep(60) elif wait_ms < 86_400_000: await broadcast_js(exec_event(Three[Selector("#emission-status")][MORPH][ @@ -1243,6 +1423,36 @@ async def get_ranking(): return {"ranking": latest.ranking, "epoch": latest.epoch} +@app.get("/api/status") +async def get_status(): + events = store.read() + discoveries = [e for e in events if isinstance(e, GitDiscovery)] + emissions = [e for e in events if isinstance(e, Emission)] + return { + **PROCESS_STATE, + "epoch": current_epoch()[0], + "openrouter_configured": bool((OPENROUTER_API_KEY or "").strip()), + "sse_clients": len(SSE_CLIENTS), + "latest_discovery": ( + { + "epoch": discoveries[-1].epoch, + "snapshot_id": discoveries[-1].snapshot_id, + "observations": len(discoveries[-1].observations), + "eligible_commits": len(discoveries[-1].commits), + } + if discoveries else None + ), + "latest_emission": ( + { + "epoch": emissions[-1].epoch, + "total_emitted": emissions[-1].total_emitted, + "ranking": emissions[-1].ranking, + } + if emissions else None + ), + } + + @app.get("/api/contributor/{github_username}") async def get_contributor(github_username: str): history = [ @@ -1306,13 +1516,6 @@ async def test_emit(): # =========================================================================== # §9. SSE — live audit stream of the pairwise voting process -# -# TODO: the /sse emission audit page needs a real SSE-driven UI. votes arrive -# incrementally during rank_commits(), and the client should show a live -# progress bar and per-vote results as they stream in. this requires a -# dedicated page that connects to /sse and updates the DOM on each event -# (council, comparing, vote, ranking, emission_complete). defer until we -# have playwright tests to cover it — the incremental rendering is fiddly. # =========================================================================== @app.get("/sse") @@ -1322,6 +1525,13 @@ async def sse_stream(request: Request): async def generate(): try: + yield _sse_event("audit", { + "id": AUDIT_SEQUENCE, + "timestamp_ms": int(time.time() * 1000), + "kind": "connection", + "message": f"Connected to epoch {current_epoch()[0]}", + **PROCESS_STATE, + }) yield exec_event(Three[Selector("#emission-status")][MORPH][ ["div#emission-status", f"Connected — epoch {current_epoch()[0]}"] ]) @@ -1334,10 +1544,15 @@ async def sse_stream(request: Request): except asyncio.TimeoutError: yield ": keepalive\n\n" finally: - SSE_CLIENTS.remove(queue) + if queue in SSE_CLIENTS: + SSE_CLIENTS.remove(queue) from starlette.responses import StreamingResponse - return StreamingResponse(generate(), media_type="text/event-stream") + return StreamingResponse( + generate(), + media_type="text/event-stream", + headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"}, + ) # =========================================================================== @@ -1359,6 +1574,7 @@ def _page(title: str, body: list) -> HTMLResponse: ["meta", {"charset": "utf-8"}], ["meta", {"name": "viewport", "content": "width=device-width, initial-scale=1"}], ["title", title], + ["style", RawContent(_WATCH_CSS)], ], ["body", body, @@ -1367,6 +1583,387 @@ def _page(title: str, body: list) -> HTMLResponse: ])) +_WATCH_CSS = """ +/* ================================================================ + ZIGGURAT — bevel-first dark theme + --spread (0→1) controls bevel depth. 0 = flat. 1 = full relief. + Light source: top-left. Shadow: bottom-right. + Platforms nest. Each level is raised. Nothing is rounded. + ================================================================ */ + +:root { + color-scheme: dark; + --spread: 1; + + --g0: #080808; + --g1: #131313; + --g2: #1c1c1c; + --g3: #252525; + --g4: #2e2e2e; + --g5: #383838; + + --hi: #5e5e5e; + --lo: #050505; + --bv: calc(var(--spread) * 4px + 1px); + --bv-lg: calc(var(--spread) * 6px + 2px); + + --signal: #f0f0f0; + --prose: #c2c2c2; + --ui: #888; + --meta: #4a4a4a; + --link: #8899ee; + --code-fg: #c8dda0; + + --font-prose: "Iowan Old Style", "Palatino Linotype", Palatino, "Book Antiqua", Georgia, serif; + --font-ui: system-ui, -apple-system, sans-serif; + --font-code: ui-monospace, "Cascadia Code", "SF Mono", Menlo, monospace; +} + +*, *::before, *::after { box-sizing: border-box; } +html, body { margin: 0; padding: 0; } + +body { + background: var(--g0); + color: var(--prose); + font-family: var(--font-ui); + font-size: 14px; + line-height: 1.6; + margin: 0 auto; + max-width: 560px; + min-height: 100vh; + padding: 0 16px 48px; +} +main { width: 100%; padding: 18px 0 48px; } + +h1, h2, h3 { + color: var(--signal); + font-size: 11px; + font-weight: bold; + letter-spacing: 0.12em; + margin: 14px 0 6px; + text-transform: uppercase; +} +a { color: var(--link); text-decoration: none; } +a:hover { color: var(--signal); } +.eyebrow { + background: var(--g2); + border: var(--bv) solid; + border-color: var(--hi) var(--lo) var(--lo) var(--hi); + color: var(--ui); + font-size: 11px; + letter-spacing: 0.12em; + padding: 4px 10px; + text-transform: uppercase; + width: fit-content; +} + +/* Every dashboard section is a raised platform. */ +.panel { + background: var(--g2); + border: var(--bv-lg) solid; + border-color: var(--hi) var(--lo) var(--lo) var(--hi); + margin: 8px 0; + padding: 10px; + width: 100%; +} +.status-row { + align-items: center; + display: flex; + flex-wrap: wrap; + gap: 8px; + justify-content: space-between; +} +#process-status { color: var(--signal); font-family: var(--font-code); font-weight: bold; } +.badge { + align-items: center; + background: var(--g3); + border: var(--bv) solid; + border-color: var(--hi) var(--lo) var(--lo) var(--hi); + color: var(--ui); + display: inline-flex; + font-size: 11px; + gap: 7px; + padding: 3px 8px; +} +.dot { background: var(--meta); height: 8px; width: 8px; } +.live .dot { background: #7acc7a; } +.warn .dot { background: #cc9955; } + +/* The progress track is inset; its signal is raised inside it. */ +.progress-shell { + background: var(--g1); + border: var(--bv-lg) solid; + border-color: var(--lo) var(--hi) var(--hi) var(--lo); + height: 58px; + margin: 14px 0 10px; + overflow: hidden; + position: relative; +} +#progress-fill { + background: var(--link); + border: var(--bv) solid; + border-color: var(--hi) var(--lo) var(--lo) var(--hi); + height: 100%; + transition: width .35s steps(8, end); + width: 0; +} +#progress-label { + color: var(--signal); + display: grid; + font-family: var(--font-code); + font-size: 18px; + font-weight: bold; + inset: 0; + place-items: center; + position: absolute; + text-shadow: 1px 1px var(--lo); +} + +.controls { align-items: center; display: flex; flex-wrap: wrap; gap: 8px; } +button { + background: var(--g5); + border: var(--bv) solid; + border-color: var(--hi) var(--lo) var(--lo) var(--hi); + color: var(--signal); + cursor: pointer; + font: inherit; + font-size: 12px; + padding: 4px 10px; +} +button:hover { background: #404040; } +button:active { + background: var(--g4); + border-color: var(--lo) var(--hi) var(--hi) var(--lo); + transform: translate(1px, 1px); +} +button:disabled { cursor: default; opacity: .4; } +.note { color: var(--meta); font-size: 11px; margin: 4px 0; } + +.feed-head { align-items: baseline; display: flex; justify-content: space-between; } +#audit-feed { + background: var(--g1); + border: var(--bv) solid; + border-color: var(--lo) var(--hi) var(--hi) var(--lo); + display: flex; + flex-direction: column; + gap: 5px; + margin-top: 8px; + padding: 6px; +} +.event { + background: var(--g3); + border: var(--bv) solid; + border-color: var(--hi) var(--lo) var(--lo) var(--hi); + display: grid; + gap: 6px; + grid-template-columns: 82px 88px 1fr; + padding: 5px 8px; +} +.event[data-kind="error"] { border-left-color: #cc5555; } +.event[data-kind="complete"], .event[data-kind="ranking"] { border-left-color: #7acc7a; } +.event[data-kind="vote"] { border-left-color: var(--link); } +.event time, .event-kind { color: var(--meta); font-family: var(--font-code); font-size: 10px; } +.event-kind { text-transform: uppercase; } +.event-message { color: var(--prose); font-family: var(--font-prose); } + +code { + background: var(--g1); + border: 2px solid; + border-color: var(--lo) var(--hi) var(--hi) var(--lo); + color: var(--code-fg); + font-family: var(--font-code); + font-size: 12px; + padding: 1px 4px; +} + +@media (max-width: 520px) { + .event { grid-template-columns: 72px 1fr; } + .event-message { grid-column: 1 / -1; } +} +""" + + +def _watch_initial_state() -> dict: + events = store.read() + feed = [] + for event_ in events[-40:]: + if isinstance(event_, GitDiscovery): + feed.append({ + "id": f"discovery-{event_.snapshot_id}", + "timestamp_ms": event_.timestamp_ms, + "kind": "discovery", + "message": ( + f"Epoch {event_.epoch}: observed {len(event_.observations)} commits; " + f"{len(event_.commits)} eligible" + ), + }) + elif isinstance(event_, Emission): + feed.append({ + "id": f"emission-{event_.epoch}", + "timestamp_ms": event_.timestamp_ms, + "kind": "complete", + "message": ( + f"Epoch {event_.epoch}: emitted {event_.total_emitted} SLG; " + f"ranking {event_.ranking}" + ), + }) + feed.extend(AUDIT_HISTORY) + return { + "process": dict(PROCESS_STATE), + "openrouter_configured": bool((OPENROUTER_API_KEY or "").strip()), + "epoch": current_epoch()[0], + "feed": feed[-200:], + } + + +_WATCH_JS = """ +const initial = __INITIAL__; +const feed = document.querySelector('#audit-feed'); +const processStatus = document.querySelector('#process-status'); +const connection = document.querySelector('#connection-status'); +const fill = document.querySelector('#progress-fill'); +const progressLabel = document.querySelector('#progress-label'); +const play = document.querySelector('#play'); +const pause = document.querySelector('#pause'); +const seen = new Set(); +let source = null; + +function setProgress(value) { + const n = Math.max(0, Math.min(100, Number(value ?? 0))); + fill.style.width = `${n}%`; + progressLabel.textContent = `${Math.round(n)}%`; + document.querySelector('.progress-shell').setAttribute('aria-valuenow', String(n)); +} + +function addEvent(event) { + const id = String(event.id); + if (seen.has(id)) return; + seen.add(id); + const row = document.createElement('div'); + row.className = 'event'; + row.dataset.kind = event.kind || 'event'; + const when = document.createElement('time'); + when.dateTime = new Date(event.timestamp_ms).toISOString(); + when.textContent = new Date(event.timestamp_ms).toLocaleTimeString(); + const kind = document.createElement('span'); + kind.className = 'event-kind'; + kind.textContent = event.kind || 'event'; + const message = document.createElement('span'); + message.className = 'event-message'; + message.textContent = event.message; + row.append(when, kind, message); + feed.prepend(row); + while (feed.children.length > 200) feed.lastElementChild.remove(); +} + +function applyState(event) { + processStatus.textContent = event.message || 'Waiting for the next epoch'; + setProgress(event.progress); + if (event.kind !== 'connection') addEvent(event); +} + +function connect() { + if (source) return; + source = new EventSource('/sse'); + connection.classList.remove('warn'); + connection.classList.add('live'); + connection.querySelector('span:last-child').textContent = 'connecting'; + play.disabled = true; + pause.disabled = false; + source.onopen = () => { + connection.querySelector('span:last-child').textContent = 'live'; + }; + source.addEventListener('audit', event => applyState(JSON.parse(event.data))); + source.onerror = () => { + connection.classList.remove('live'); + connection.classList.add('warn'); + connection.querySelector('span:last-child').textContent = 'reconnecting'; + }; +} + +function disconnect() { + if (source) source.close(); + source = null; + connection.classList.remove('live'); + connection.classList.add('warn'); + connection.querySelector('span:last-child').textContent = 'paused locally'; + play.disabled = false; + pause.disabled = true; +} + +play.addEventListener('click', connect); +pause.addEventListener('click', disconnect); +initial.feed.forEach(addEvent); +processStatus.textContent = initial.process.message; +setProgress(initial.process.progress); +connect(); +""" + + +@app.get("/watch") +async def watch(): + initial = json.dumps( + _watch_initial_state(), separators=(",", ":") + ).replace(" 0") - ;; 9. SSE connects and sends initial event + ;; 9. watch UI exposes progress, controls, readiness, and live SSE + (println "\nchecking /watch UI…") + (bind watch-html (slurp (str base-url "/watch"))) + (assert! (str/includes? watch-html "role=\"progressbar\"") + "watch page has progress bar") + (assert! (str/includes? watch-html "id=\"play\"") + "watch page has play control") + (assert! (str/includes? watch-html "id=\"pause\"") + "watch page has pause control") + (assert! (str/includes? watch-html "OpenRouter configured") + "watch page reports council readiness") + (bind status-resp (get-json base-url "/api/status")) + (assert! (true? (:openrouter_configured status-resp)) + "status API reports OpenRouter configuration") + + ;; 10. SSE connects and sends initial event (println "\nchecking /sse initial event…") (bind sse-events (read-sse-events (str base-url "/sse") 1 5000)) (assert! (= 1 (count sse-events)) "received 1 SSE event") (assert! (not (str/blank? (first sse-events))) "initial SSE event contains executable audit data") - ;; 10. POST /test/emit — full ranking pipeline hits mocks + ;; 11. POST /test/emit — full ranking pipeline hits mocks (println "\ntriggering /test/emit (epoch 1)…") (bind emit-resp (post-json! base-url "/test/emit")) (assert! (= "emission" (:type emit-resp)) "emit response type is emission") @@ -503,7 +518,7 @@ (bind rank-after (get-json base-url "/api/ranking")) (assert! (= 1 (:epoch rank-after)) "latest ranking is epoch 1") - ;; 11. kill and restart — prove replay determinism + ;; 12. kill and restart — prove replay determinism (println "\nkilling server for replay test…") (.destroyForcibly (:proc server)) (deref server) diff --git a/tests/test_git_discovery.py b/tests/test_git_discovery.py index 0dd31bc42a19bc8c59842dc61f193c595c474659..5a9e167e16ad2ffb25988af85820f6f19e8910cd 100644 --- a/tests/test_git_discovery.py +++ b/tests/test_git_discovery.py @@ -393,7 +393,9 @@ def test_empty_epoch_records_zero_emission_without_burning_pool( monkeypatch.setattr(c, "store", c.JsonlStore(discovery_config / "ledger.jsonl")) async def discover(_epoch, _boundary): - return SimpleNamespace(commits=[], snapshot_id="empty-snapshot") + return SimpleNamespace( + observations=[], commits=[], snapshot_id="empty-snapshot" + ) async def rank(_commits): return {}, [] @@ -412,7 +414,11 @@ def test_emission_distribution_sums_exactly_to_total( monkeypatch.setattr(c, "store", c.JsonlStore(discovery_config / "ledger.jsonl")) async def discover(_epoch, _boundary): - return SimpleNamespace(commits=[{"x": 1}], snapshot_id="ranked-snapshot") + return SimpleNamespace( + observations=[{"x": 1}], + commits=[{"x": 1}], + snapshot_id="ranked-snapshot", + ) async def rank(_commits): return { @@ -454,6 +460,7 @@ def test_any_council_failure_aborts_ranking(monkeypatch): monkeypatch.setattr(c, "fetch_top_models", models) monkeypatch.setattr(c, "llm_pairwise_compare", compare) + monkeypatch.setattr(c, "OPENROUTER_API_KEY", "test-key") commits = [ { "contributor": contributor, @@ -465,3 +472,54 @@ def test_any_council_failure_aborts_ranking(monkeypatch): ] with pytest.raises(RuntimeError, match="council model failed"): asyncio.run(c.rank_commits(commits)) + + +def test_contested_ranking_requires_openrouter_key(monkeypatch): + monkeypatch.setattr(c, "OPENROUTER_API_KEY", "") + commits = [ + { + "contributor": contributor, + "oid": "sha1:" + char * 40, + "message": contributor, + "patch": "patch", + } + for contributor, char in [("alice", "a"), ("bob", "b")] + ] + with pytest.raises(RuntimeError, match="OPENROUTER_API_KEY"): + asyncio.run(c.rank_commits(commits)) + + +def test_watch_page_has_live_controls_progress_and_key_warning( + discovery_config, monkeypatch +): + monkeypatch.setattr(c, "store", c.JsonlStore(discovery_config / "ledger.jsonl")) + monkeypatch.setattr(c, "OPENROUTER_API_KEY", "") + monkeypatch.setattr(c, "current_epoch", lambda: (3, 0, 1)) + response = asyncio.run(c.watch()) + html = response.body.decode() + assert 'role="progressbar"' in html + assert 'id="play"' in html + assert 'id="pause"' in html + assert "new EventSource('/sse')" in html + assert "OpenRouter key missing" in html + + +def test_audit_events_are_json_sse_and_update_process_state(monkeypatch): + clients = [] + history = [] + monkeypatch.setattr(c, "SSE_CLIENTS", clients) + monkeypatch.setattr(c, "AUDIT_HISTORY", history) + queue = asyncio.Queue() + clients.append(queue) + + async def emit(): + event = await c.broadcast_audit( + "progress", "halfway", progress=50, phase="ranking" + ) + return event, await queue.get() + + event, wire = asyncio.run(emit()) + assert event["progress"] == 50 + assert event["phase"] == "ranking" + assert wire.startswith("event: audit\ndata: {") + assert '"message":"halfway"' in wire Side B — contributor: tommy-mor Side B — commit message: [3b3d5873] item refactor Side B — unified diff (full patch): diff --git a/plan.md b/plan.md deleted file mode 100644 index 00d6867a1e0ed144a16a020ea037f685ce646c73..0000000000000000000000000000000000000000 --- a/plan.md +++ /dev/null @@ -1,155 +0,0 @@ -# Plan: `ItemId` + `RouteContext` (identity vs hrefs) - -This document is for **the next agent** to continue the refactor without re-deriving context from chat. It supersedes ad-hoc notes: treat it as the checklist of record until the work lands and this file is deleted or trimmed. - -## Goal - -- **Identity** (what lives in the reducer graph, votes, indexes) becomes a **structural `ItemId` enum** in `slug-types`, not a canonical `String` / `CanonicalItemUrl` newtype. -- **Presentation** (tilde / dash display, breadcrumbs) derives from `ItemId` via explicit methods, not string stripping. -- **Routing** (browser `href`s for public vs room) goes through **`RouteContext`** (started in `server/src/html/routing.rs`) so Maud/handlers do not stitch `/r/…` vs `/~` ad hoc. - -**Non-goals for v1 of the migration:** backward-compatible JSONL or dual-read of old canonical strings in the event log (project has accepted breaking changes). If you reintroduce compat, document it here. - -## Current state (as of this plan) - -- **`CanonicalItemUrl`** (`types/src/paths.rs`): newtype around `String`; `parse` / `parent` / `display_path` / `tilde_tail` / etc. Reducer `ContentState`, `VoteData`, ranking, RPC, search, garden, breadcrumbs all use it or `String` keys derived from it. -- **`ThreadNav`** (`server/src/html/forum/nav.rs`): encodes scope prefixes for threads and garden URLs; **`RouteContext`** now wraps `ThreadNav` (`server/src/html/routing.rs`, re-exported from `server/src/html/mod.rs`) but **most HTML still takes `&ThreadNav` directly** — migration incomplete. -- **URL normalization** lives in `types/src/url_normalize.rs` + `canonicalize_item` / `finalize_external_identity_url` in `paths.rs` (YouTube, sorted query params, room path `room_route_segment` in `paths.rs`). -- **Room HTTP paths** are `/r/{short}{slug}` (fused segment); wire **`room_id`** remains `short/slug` for RPC/events. - -## Target architecture - -### `ItemId` (types) - -Suggested shape (adjust after profiling `Ord` / `Hash` / serde size): - -```text -ItemId::Root — tilde ontology root (today `SLUG_TILDE_ONTOLOGY_ROOT`) -ItemId::Local { segments } — slug.social ~/… path as Vec (lowercase segments, non-empty for non-root) -ItemId::External { url: Url } — normalized `url::Url` (crate `url` already in `slug-types`) -``` - -**API surface (minimum):** - -- `ItemId::parse(&str) -> Option` — single entry from DSL / user input / legacy wire (internally may call `canonicalize_item` + structured split). -- `ItemId::to_wire_url(&self) -> String` — only for **external** boundaries if needed (HTTP fetch, rare assertions); avoid using as the primary key once maps use `ItemId`. -- `parent`, `display_path`, `tilde_tail` / `tilde_http_tail`, `tilde_segments`, `last_segment`, `normalized_storage` — port from `CanonicalItemUrl`. -- **`Ord` + `Hash` + `Eq`** stable for `BTreeSet` / `HashMap` (see `write_actor` scope-rank snapshots). -- **`Serialize` / `Deserialize`** — decide **tagged JSON** for any persisted or API-carried structs (e.g. `VoteData` in tests). If RPC must stay stringy for clients, use a **DTO layer** that converts `ItemId` ↔ wire at the boundary only. - -**Remove:** `CanonicalItemUrl` type and all `path_types::CanonicalItemUrl` / `slug_types::paths::CanonicalItemUrl` exports once call sites are migrated. **`Borrow`** on the old newtype goes away; update `nav!` / any code that assumed map keys borrowed as `str`. - -### `RouteContext` (server HTML) - -- **File:** `server/src/html/routing.rs` — **`RouteContext(ThreadNav)`** with `item_href`, `item_href_raw`, `thread_url`, `garden_root_url`, `room_url`, `From`/`Into` `ThreadNav`. -- **Direction:** new code and refactored Maud should take **`&RouteContext`** (or owned where appropriate) instead of `&ThreadNav` when building links. Long term, **`item_href(&ItemId)`** should not parse strings — it should pattern-match `ItemId` and append tilde tail or `/-/…` external tail using the same rules as today’s `ThreadNav::garden_item_url`. - -### Axum / garden routes - -- **No** single catch-all route (explicit decision): keep the existing router layout in `server/src/lib.rs`. -- Room routes stay **`/r/:room_key/...`** with `room_key` fused; parsing via `slug_types::room_id_from_route_segment` / `room_route_segment` in `paths.rs`. - -## Phased execution (recommended order) - -### Phase 0 — Preconditions (quick) - -1. Read **`AGENTS.md`** (UI contract, durability matrix, `RpcCommand` vs `HtmlUiAction`). -2. Run **`cargo test --workspace`** and **`./scripts/clj-test.sh`** on clean `main` before large diffs; repeat after each phase. - -### Phase 1 — `ItemId` in `slug-types` (no server yet) - -1. Add **`ItemId`** (new file e.g. `types/src/item_id.rs` **or** inline at bottom of `paths.rs` — see **Module cycle** below). -2. Implement **`ItemId::parse`** using existing **`canonicalize_item`** + normalization; port **`CanonicalItemUrl`** methods to **`ItemId`** with tests ported from `paths.rs` `#[cfg(test)] mod tests`. -3. **`GardenItemUrl::from_stored(&ItemId, room_wire)`** (and thread helpers) — build absolute hrefs from structure, not from re-parsing a canonical string. -4. **`TildeHttpPathTail::to_item_id`** (rename from `to_canonical`) / **`tilde_http_path_to_item_id`**. -5. **`TildeOntologyPath::from_stored(&ItemId)`**. -6. Export **`ItemId`** from **`types/src/lib.rs`**; update **`server/src/path_types.rs`** re-exports. -7. **Delete `CanonicalItemUrl`** and fix all **in-crate** references in `types` only until `cargo test` passes for `slug-types`. - -**Module cycle trap:** `item_id.rs` must not `use crate::paths::{...}` if `paths.rs` also imports `ItemId` for `GardenItemUrl` in the same module. **Fix one of:** - -- **A)** Put `ItemId` **inside `paths.rs`** below `canonicalize_item` / helpers (simplest, large file), or -- **B)** Split **`canonicalize_item`** (+ dash host helpers + `finalize_external_identity_url`) into **`types/src/item_wire.rs`**, then `paths.rs` + `item_id.rs` both depend on `item_wire` only (cleaner, more files). - -### Phase 2 — Reducer + ranking (server core) - -1. **`server/src/reducer.rs`**: `ContentState` / `GroupState` / **`VoteData`** — replace **`CanonicalItemUrl`** with **`ItemId`** on all maps, sets, deques, vectors. -2. **`apply_vote`**: normalize `a`/`b` via **`ItemId::parse`** or **`ItemId`**-aware logic (remove string round-trip). -3. **`apply_ingest_to_content`**: **`dsl`** still yields strings for item titles in statements; normalize to **`ItemId`** at ingest boundary via **`ItemId::parse`** once per item. -4. **`server/src/ranking.rs`**, **`server/src/scope_rank.rs`**, **`server/src/api/write_actor.rs`** (including **`BTreeSet`** ordering), **`server/src/api/validate.rs`**, **`server/src/api/helpers.rs`** — propagate **`ItemId`**. -5. **`server/tests/basic.rs`** and any reducer tests constructing **`VoteData`** — use **`ItemId::parse(...).unwrap()`** or helpers. - -### Phase 3 — RPC + search + external resolver - -1. **`server/src/api/rpc.rs`**: rank/pair/matchup/search payloads; today many paths use **`GardenItemUrl::from_storage_str(item.as_str(), …)`** — switch to **`ItemId`** + **`GardenItemUrl::from_stored(&item_id, …)`** (or equivalent). -2. **`server/src/html/search.rs`**: scoring uses item path strings — derive from **`ItemId::display_path`** / **`to_wire_url`** only at the scoring boundary if needed. -3. **`server/src/external_resolver.rs`**: take **`&ItemId`** or **`ItemId::external_url()`** instead of **`&CanonicalItemUrl`**. - -### Phase 4 — HTML / Maud - -1. **`ThreadNav::garden_item_url`**: overload or replace with **`garden_item_href(&self, item: &ItemId)`** (no `CanonicalItemUrl::parse` inside). -2. **`RouteContext`**: extend **`item_href(&ItemId)`**; migrate call sites from **`ThreadNav`** to **`RouteContext`** where only link-building is needed (keep **`ThreadNav`** where scope / auth helpers need the full struct). -3. **`server/src/html/garden.rs`**, **`breadcrumb_path.rs`**, **`forum/*`**, **`editor.rs`**: replace **`CanonicalItemUrl`** with **`ItemId`**; breadcrumbs should walk **`ItemId::parent`** without string `rsplit`. -4. **`types` JSON types** (`RankRow`, etc.): decide whether **`GardenItemUrl`** stays string for JSON or becomes a structured field; keep **one** wire format for the public API. - -### Phase 5 — Cleanup + docs - -1. Remove dead **`canonical_path`** / **`breadcrumb_path`** string logic if fully superseded. -2. Update **`AGENTS.md`** if durability, `POST /ui`, or command surfaces change. -3. Delete or shrink **`plan.md`** when done. - -## File / symbol checklist (non-exhaustive — grep-driven) - -Run periodically: - -```bash -rg "CanonicalItemUrl" -g'*.rs' -rg "path_types::CanonicalItemUrl" -g'*.rs' -rg "tilde_http_path_to_canonical" -g'*.rs' -``` - -**High-touch files (from prior exploration):** - -| Area | Files | -|------|--------| -| Types | `types/src/paths.rs`, `types/src/lib.rs`, `types/src/url_normalize.rs`, (optional) `types/src/item_id.rs`, `types/src/item_wire.rs` | -| Server re-exports | `server/src/path_types.rs`, `server/src/canonical_path.rs` | -| Reducer / ingest | `server/src/reducer.rs`, `server/src/dsl.rs` (parse output types if changed) | -| Ranking | `server/src/ranking.rs`, `server/src/scope_rank.rs` | -| Writer / RPC | `server/src/api/write_actor.rs`, `server/src/api/rpc.rs`, `server/src/api/helpers.rs`, `server/src/api/validate.rs` | -| HTML | `server/src/html/garden.rs`, `server/src/html/breadcrumb_path.rs`, `server/src/html/forum/nav.rs`, `server/src/html/routing.rs`, `server/src/html/search.rs`, `server/src/html/editor.rs`, `server/src/html/forum/ingest.rs`, … | -| Tests | `server/tests/basic.rs`, `server/tests/integration.rs`, `types/src/paths.rs` tests, Clojure under `test/` if URLs/assertions mention canonical shapes | - -## Events / JSONL - -- **`Ingest`** events store **`raw` DSL** only — no change required for item identity inside the event. -- If any future event type stores item ids as strings, migrate to **structured `ItemId` serde** or accept string only at the event boundary with immediate parse into **`ItemId`** on `apply_event`. - -## `nav!` macro (`server/src/paths.rs`) - -- Macros use **`keypath($key)`** with **`.clone()`** — **`ItemId`** must be **`Clone`** (already for enums). Remove any reliance on **`Borrow`** for map keys. - -## Testing gate - -After each phase: - -```bash -cargo test --workspace -./scripts/clj-test.sh -``` - -## Risks / gotchas - -1. **`Ord` on `ItemId`**: must match prior **`CanonicalItemUrl`** / `String` ordering wherever **`BTreeSet`** is used (e.g. deterministic scope-rank snapshots in **`write_actor`**). -2. **External `ItemId`**: **`Url`** equality / hashing — normalization is already centralized in **`url_normalize`**; ensure **`ItemId::parse`** always inserts normalized **`Url`** into **`External`**. -3. **Fake parent URLs** in garden (e.g. **`https://.`** for external root ranking): find all **`parse("https://.")`** style hacks and express as **`ItemId`** or a dedicated sentinel. -4. **Serde**: tests and any RPC clients that snapshot JSON may need expectation updates if **`VoteData`** shape changes. - -## Optional follow-ups (not blocking `ItemId`) - -- More **domain normalizers** in **`url_normalize.rs`** (e.g. `music.youtube.com`, Spotify, etc.). -- **Room wire** vs **HTTP segment** helpers already in **`paths.rs`** (`ROOM_SHORT_ID_LEN`, `room_route_segment`, `room_id_from_route_segment`). - ---- - -**End state criteria:** `rg CanonicalItemUrl` returns nothing; reducer maps use **`ItemId`**; HTML link generation for items goes through **`RouteContext` + `ItemId`**; tests and Kaocha green. diff --git a/server/src/api/helpers.rs b/server/src/api/helpers.rs index 1b291db83df7364a026f2e147e0a29a70a399371..8cd23a02fa5219d6aa375e766e6b2bd2c7bb7dfb 100644 --- a/server/src/api/helpers.rs +++ b/server/src/api/helpers.rs @@ -4,7 +4,8 @@ use axum::{ Json, }; use sha2::{Digest, Sha256}; -use slug_types::paths::{CanonicalItemUrl, GardenItemUrl}; +use slug_types::paths::GardenItemUrl; +use slug_types::ItemId; use slug_types::*; use std::collections::HashMap; @@ -31,12 +32,12 @@ pub fn now_ms() -> i64 { } /// Resolve DSL/user input to a stored canonical item id. -pub fn resolve_item(item: &str) -> Result { +pub fn resolve_item(item: &str) -> Result { let canonical = canonicalize_item(item); if canonical.is_empty() { return Err(format!("empty item path: `{}`", item)); } - Ok(CanonicalItemUrl(canonical)) + ItemId::parse(&canonical).ok_or_else(|| format!("invalid item path: `{}`", item)) } pub fn parse_parent_specs(parent: Option<&String>) -> Vec { @@ -94,7 +95,7 @@ pub fn paginate_rankings( (out_components, out_unranked) } -pub fn pick_random_distinct_canonical(items: &[CanonicalItemUrl]) -> Option<(CanonicalItemUrl, CanonicalItemUrl)> { +pub fn pick_random_distinct_canonical(items: &[ItemId]) -> Option<(ItemId, ItemId)> { use rand::seq::SliceRandom; if items.len() < 2 { return None; @@ -115,15 +116,15 @@ pub fn pick_random_distinct_canonical(items: &[CanonicalItemUrl]) -> Option<(Can } pub fn is_pair_voted(group: &crate::reducer::GroupState, a: &str, b: &str) -> bool { - let a_key = CanonicalItemUrl(a.to_string()); - let b_key = CanonicalItemUrl(b.to_string()); + let a_key = ItemId::parse(a).unwrap_or_else(|| ItemId::opaque(a.to_string())); + let b_key = ItemId::parse(b).unwrap_or_else(|| ItemId::opaque(b.to_string())); let Some(&a_idx) = group.item_to_idx.get(&a_key) else { return false; }; let Some(&b_idx) = group.item_to_idx.get(&b_key) else { return false; }; let (i, j) = if a_idx < b_idx { (a_idx, b_idx) } else { (b_idx, a_idx) }; group.voted_pairs.contains(&(i, j)) } -pub fn compute_connectivity_stats(group: &crate::reducer::GroupState, pool: &[CanonicalItemUrl]) -> ConnectivityStats { +pub fn compute_connectivity_stats(group: &crate::reducer::GroupState, pool: &[ItemId]) -> ConnectivityStats { let n = pool.len(); let global_idxs: Vec> = pool diff --git a/server/src/api/rpc.rs b/server/src/api/rpc.rs index 0ff2d701dd1e8661f58abd55672cd91280e9491e..079d96eb6f1d717c203b1d4aec09f3916384b919 100644 --- a/server/src/api/rpc.rs +++ b/server/src/api/rpc.rs @@ -17,7 +17,7 @@ use crate::{ dsl, events::{Event, Ingest, ThreadCapability}, identity::{parse_agent, parse_username}, - path_types::CanonicalItemUrl, + path_types::ItemId, ranking::{connected_components_from_voted_pairs, ranked_items_subset}, reducer::{scope_from_room_wire, ReducerState, ScopeId}, state::{AppState, InviteState}, @@ -150,7 +150,7 @@ fn build_rank_response_for_content( if !is_global && !specs.is_empty() { let none_exist = specs.iter().all(|spec| { - let Some(canon) = CanonicalItemUrl::parse(spec) else { return true }; + let Some(canon) = ItemId::parse(spec) else { return true }; !content.items.contains(&canon) && !content.item_children.contains_key(&canon) }); if none_exist { @@ -163,10 +163,10 @@ fn build_rank_response_for_content( let depth = depth.max(1); let rankings = if is_global { - let all_items: Vec = content.items.iter().cloned().collect(); + let all_items: Vec = content.items.iter().cloned().collect(); crate::scope_rank::build_rankings_for_item_set(content, &all_items) } else if specs.is_empty() { - crate::scope_rank::build_children_rankings(content, &CanonicalItemUrl::ontology_root()) + crate::scope_rank::build_children_rankings(content, &ItemId::ontology_root()) } else if depth > 1 { let items = crate::scope_rank::resolve_scope_recursive(content, &specs, depth); crate::scope_rank::build_rankings_for_item_set(content, &items) @@ -359,8 +359,8 @@ async fn rpc_check( let mut simulated = { reduced_arc.read().await.clone() }; simulated.apply_event(event); - let voted_parents: Vec = { - let mut parents: HashSet = HashSet::new(); + let voted_parents: Vec = { + let mut parents: HashSet = HashSet::new(); for s in &v.doc.statements { if let dsl::Stmt::Vote { item1, item2, .. } = s { if let (Ok(a), Ok(b)) = (resolve_item(item1), resolve_item(item2)) { @@ -369,7 +369,7 @@ async fn rpc_check( } } } - let mut out: Vec = parents.into_iter().collect(); + let mut out: Vec = parents.into_iter().collect(); out.sort(); out }; @@ -695,7 +695,7 @@ fn rpc_search(reduced: &ReducerState, q: &str, limit: usize, principal: Option<& async fn rpc_get_pair(state: &AppState, room: String, parent_path: String) -> Result { let scope = scope_from_room_wire(&room); let reduced_arc = state.reduced.clone(); - let pool: Vec = { + let pool: Vec = { let reduced = reduced_arc.read().await; let content = content_for_room(&reduced, &room); let tmp = if parent_path.trim().is_empty() { @@ -716,7 +716,7 @@ async fn rpc_get_pair(state: &AppState, room: String, parent_path: String) -> Re Some("add items via ingest".into()), )); } - let selected: Option<(CanonicalItemUrl, CanonicalItemUrl)> = { + let selected: Option<(ItemId, ItemId)> = { let mut reduced = reduced_arc.write().await; let content = reduced.content.entry(scope.clone()).or_default(); let group = &mut content.ranking_group; @@ -729,16 +729,16 @@ async fn rpc_get_pair(state: &AppState, room: String, parent_path: String) -> Re .filter_map(|it| group.item_to_idx.get(it).copied()) .collect(); let ranked = ranked_items_subset(group, &idxs, 10000, 1e-8); - let ranked_set: HashSet = ranked.iter().map(|r| r.item.clone()).collect(); - let unsorted: Vec = pool + let ranked_set: HashSet = ranked.iter().map(|r| r.item.clone()).collect(); + let unsorted: Vec = pool .iter() .filter(|it| !ranked_set.contains(*it)) .cloned() .collect(); - let mut pick: Option<(CanonicalItemUrl, CanonicalItemUrl)> = None; + let mut pick: Option<(ItemId, ItemId)> = None; if !unsorted.is_empty() { if let Some(left) = unsorted.choose(&mut rng).cloned() { - let mut candidates: Vec = if !ranked.is_empty() { + let mut candidates: Vec = if !ranked.is_empty() { ranked.iter().map(|r| r.item.clone()).collect() } else { pool.clone() @@ -865,7 +865,7 @@ pub async fn handle_rpc_batch( } else { let content = content_for_room(&reduced, &room); let item_str = canonicalize_item(&item_path); - let item = CanonicalItemUrl(item_str.clone()); + let item = ItemId::parse(&item_str).unwrap_or_else(|| ItemId::opaque(item_str.clone())); if !content.items.contains(&item) { line_err( "item not found", @@ -1212,7 +1212,7 @@ pub async fn handle_rpc_batch( } let ranked_total = ranked.len(); - let mut unranked: Vec = content + let mut unranked: Vec = content .items .iter() .filter(|it| !group.item_to_idx.contains_key(*it)) @@ -1263,7 +1263,7 @@ pub async fn handle_rpc_batch( } else { let content = content_for_room(&reduced, &room); let item_str = canonicalize_item(&item_path); - let item = CanonicalItemUrl(item_str.clone()); + let item = ItemId::parse(&item_str).unwrap_or_else(|| ItemId::opaque(item_str.clone())); let limit = limit.unwrap_or(50).clamp(1, 200); if !content.items.contains(&item) { line_err( @@ -1304,7 +1304,7 @@ pub async fn handle_rpc_batch( let content = content_for_room(&reduced, &room); let scope = scope_from_room_wire(&room); let item_str = canonicalize_item(&item_path); - let item = CanonicalItemUrl(item_str.clone()); + let item = ItemId::parse(&item_str).unwrap_or_else(|| ItemId::opaque(item_str.clone())); let entries = content.rank_history.get(&item).cloned().unwrap_or_default(); let history: Vec = entries.iter().map(|e| { let caused_by: Vec = reduced.ingests_by_id.get(&e.post_id) @@ -1361,7 +1361,11 @@ pub async fn handle_rpc_batch( line_err(e, h) } else { let content = content_for_room(&reduced, &room); - let parents: HashSet<&str> = content.item_children.keys().map(|s| s.as_str()).collect(); + let parents: HashSet = content + .item_children + .keys() + .map(|k| k.to_storage_string()) + .collect(); let mut paths: Vec = content .items .iter() @@ -1380,11 +1384,11 @@ pub async fn handle_rpc_batch( let content = content_for_room(&reduced, &room); let out: Vec = content .item_children - .get(&CanonicalItemUrl::ontology_root()) + .get(&ItemId::ontology_root()) .map(|roots| { let mut v: Vec = roots.iter() .map(|path| { - let children = content.item_children.get(path.as_str()).map(|s| s.len()).unwrap_or(0); + let children = content.item_children.get(path).map(|s| s.len()).unwrap_or(0); PathSummary { path: TildeOntologyPath::from_stored(path), children, diff --git a/server/src/api/validate.rs b/server/src/api/validate.rs index 3c7aa80fd485cf247231fe5c5c7e6fea62f22534..3f657a88ef9328a31e3e29c3bc3bd32bec6fd7a1 100644 --- a/server/src/api/validate.rs +++ b/server/src/api/validate.rs @@ -4,7 +4,7 @@ use std::collections::HashSet; use crate::{ canonical_path::canonicalize_tag, dsl, - path_types::CanonicalItemUrl, + path_types::ItemId, reducer::{ReducerState, ScopeId}, }; use slug_types::paths::GardenItemUrl; @@ -32,11 +32,11 @@ pub fn validate_ingest_document( ScopeId::Public => None, _ => reduced.content_for_scope(scope), }; - let item_exists = |key: &CanonicalItemUrl| { + let item_exists = |key: &ItemId| { scoped_content.map(|c| c.items.contains(key)).unwrap_or(false) || public_content.items.contains(key) }; - let body_exists = |key: &CanonicalItemUrl| { + let body_exists = |key: &ItemId| { scoped_content.map(|c| c.item_bodies.contains_key(key)).unwrap_or(false) || public_content.item_bodies.contains_key(key) }; @@ -52,7 +52,7 @@ pub fn validate_ingest_document( }; let ts = super::helpers::now_ms(); - let mut defined_in_doc: HashSet = HashSet::new(); + let mut defined_in_doc: HashSet = HashSet::new(); for s in &doc.statements { match s { diff --git a/server/src/api/write_actor.rs b/server/src/api/write_actor.rs index f9c3b8bd3fbf8fcb9c035e1a1572fef0b08fa8a9..3e96b015def54921962f74b4cd884808ff6c3bcc 100644 --- a/server/src/api/write_actor.rs +++ b/server/src/api/write_actor.rs @@ -10,7 +10,7 @@ use crate::{ events::{AgentBound, Event, GrantAdded, Ingest, PostRedacted, RoomDeleted, UserRegistered}, html::JsBuilder, identity::parse_agent, - path_types::CanonicalItemUrl, + path_types::ItemId, reducer::{scope_from_room_wire, ReducerState, ScopeId}, state::AppState, write_cmd::WriteCmd, @@ -90,7 +90,7 @@ async fn broadcast_web_refresh(state: &AppState, room_key: &str, thread_id: &str } fn compute_scope_rank_changes( - parent: &CanonicalItemUrl, + parent: &ItemId, before: &crate::scope_rank::ChildrenRankings, after: &crate::scope_rank::ChildrenRankings, room_wire: &str, @@ -99,7 +99,7 @@ fn compute_scope_rank_changes( use std::collections::BTreeSet; fn build_positions( rankings: &crate::scope_rank::ChildrenRankings, - ) -> HashMap> { + ) -> HashMap> { let mut map = HashMap::new(); for comp in &rankings.component_rankings { let total = comp.ranked.len(); @@ -116,7 +116,7 @@ fn compute_scope_rank_changes( let before_pos = build_positions(before); let after_pos = build_positions(after); - let all_items: BTreeSet = before_pos + let all_items: BTreeSet = before_pos .keys() .cloned() .chain(after_pos.keys().cloned()) @@ -287,8 +287,8 @@ pub async fn writer_actor(mut rx: mpsc::Receiver, state: AppState) { .map(|d| reduced.agent_bindings.get(d).is_none()) .unwrap_or(false); - let voted_parent_scopes: Vec = { - let mut parents: HashSet = HashSet::new(); + let voted_parent_scopes: Vec = { + let mut parents: HashSet = HashSet::new(); for s in &v.doc.statements { if let dsl::Stmt::Vote { item1, item2, .. } = s { if let (Ok(a), Ok(b)) = (resolve_item(item1), resolve_item(item2)) { @@ -301,12 +301,12 @@ pub async fn writer_actor(mut rx: mpsc::Receiver, state: AppState) { } } } - let mut out: Vec = parents.into_iter().collect(); + let mut out: Vec = parents.into_iter().collect(); out.sort(); out }; - let pre_rankings: HashMap = + let pre_rankings: HashMap = if !voted_parent_scopes.is_empty() { let content = content_for_room(&reduced, &room_key); voted_parent_scopes diff --git a/server/src/external_resolver.rs b/server/src/external_resolver.rs index 87409c4cab5f1930d2cd075f3417fa824fd791c7..6f5c982a627f9f620f0e3be0ba0cb92a3d7d4bb7 100644 --- a/server/src/external_resolver.rs +++ b/server/src/external_resolver.rs @@ -1,6 +1,6 @@ use async_trait::async_trait; -use crate::path_types::CanonicalItemUrl; +use crate::path_types::ItemId; #[async_trait] pub trait ExternalResolver: Send + Sync { @@ -11,7 +11,7 @@ pub trait ExternalResolver: Send + Sync { fn normalize(&self, path: &str) -> String; /// Fetches body when missing; GitHub hook lands here in a follow-up. - async fn fetch_body(&self, canonical_url: &CanonicalItemUrl) -> Result; + async fn fetch_body(&self, canonical_url: &ItemId) -> Result; } /// Placeholder until domain-specific resolvers exist. @@ -27,7 +27,7 @@ impl ExternalResolver for DefaultExternalResolver { path.to_string() } - async fn fetch_body(&self, _canonical_url: &CanonicalItemUrl) -> Result { + async fn fetch_body(&self, _canonical_url: &ItemId) -> Result { Err("external fetch not implemented".to_string()) } } diff --git a/server/src/html/breadcrumb_path.rs b/server/src/html/breadcrumb_path.rs index c8a3937923161a6ff248bd77e87eff0dc6fe9ab0..5743a98753f579c70e10961278469afd7cb9ddcf 100644 --- a/server/src/html/breadcrumb_path.rs +++ b/server/src/html/breadcrumb_path.rs @@ -1,8 +1,8 @@ -use crate::path_types::{tilde_http_path_to_canonical, CanonicalItemUrl}; +use crate::path_types::{tilde_http_path_to_item_id, ItemId}; /// Semantic view of an ontology path for rendering and routing decisions. pub(super) struct OntologyPath { - canonical: CanonicalItemUrl, + canonical: ItemId, /// Breadcrumb segments: for `~/a/b` this is `["a", "b"]` (leading `~` rendered separately). segments: Vec, } @@ -11,11 +11,11 @@ impl OntologyPath { /// Path is the `*path` segment from `/~/*path` (e.g. `topic/a`). Always treat it as under `~/` /// so it canonicalizes to `https://slug.social/~/…`, not the non-tilde site path. pub(super) fn from_input(path: &str) -> Self { - let canonical = tilde_http_path_to_canonical(path); + let canonical = tilde_http_path_to_item_id(path); Self::from_canonical(canonical) } - pub(super) fn from_canonical(canonical: CanonicalItemUrl) -> Self { + pub(super) fn from_canonical(canonical: ItemId) -> Self { // tilde_segments() returns ["~", "a", "b"] but bc_path() renders "~" itself, // so we skip the leading "~" segment here. let segments = canonical @@ -28,7 +28,7 @@ impl OntologyPath { } pub(super) fn root() -> Self { - Self::from_canonical(CanonicalItemUrl::ontology_root()) + Self::from_canonical(ItemId::ontology_root()) } pub(super) fn is_root(&self) -> bool { @@ -56,7 +56,7 @@ impl OntologyPath { /// External `https://host/…` items addressed as `/-/host/…` in the URL bar. pub(super) struct ExternalOntologyPath { - canonical: CanonicalItemUrl, + canonical: ItemId, /// e.g. `["github.com", "org", "repo", "issues"]` segments: Vec, } @@ -71,13 +71,13 @@ impl ExternalOntologyPath { } else { format!("-/{}", p.trim_start_matches('/')) }; - let Some(canonical) = CanonicalItemUrl::parse(&raw) else { - return Self::from_canonical(CanonicalItemUrl("https://.".to_string())); + let Some(canonical) = ItemId::parse(&raw) else { + return Self::from_canonical(ItemId::opaque("https://.".to_string())); }; Self::from_canonical(canonical) } - pub(super) fn from_canonical(canonical: CanonicalItemUrl) -> Self { + pub(super) fn from_canonical(canonical: ItemId) -> Self { let s = canonical.as_str(); let rest = s .strip_prefix("https://") diff --git a/server/src/html/editor.rs b/server/src/html/editor.rs index 80657ac0df9b4bb67e4cdacca296f70267840bd1..c76417acd7a164fda23537cd4e7fb9f1f1dcd895 100644 --- a/server/src/html/editor.rs +++ b/server/src/html/editor.rs @@ -97,7 +97,7 @@ pub async fn editor_check( simulated.apply_event(event); // Collect voted parent scopes. - let voted_parents: Vec = { + let voted_parents: Vec = { let mut parents = std::collections::HashSet::new(); for s in &v.doc.statements { if let crate::dsl::Stmt::Vote { item1, item2, .. } = s { @@ -107,7 +107,7 @@ pub async fn editor_check( } } } - let mut out: Vec = parents.into_iter().collect(); + let mut out: Vec = parents.into_iter().collect(); out.sort(); out }; diff --git a/server/src/html/forum/nav.rs b/server/src/html/forum/nav.rs index 48fe11e46731670874ff8b6b05baa6f09ae0b7e4..7ca5b491f38104ec8c81db43a95c59da00abd8c4 100644 --- a/server/src/html/forum/nav.rs +++ b/server/src/html/forum/nav.rs @@ -52,9 +52,14 @@ impl ThreadNav { } pub(crate) fn garden_item_url(&self, item: &str) -> String { - let Some(c) = crate::path_types::CanonicalItemUrl::parse(item) else { + let Some(c) = crate::path_types::ItemId::parse(item) else { return format!("{}/{}", self.garden_path_prefix, canonicalize_item(item)); }; + self.garden_item_href(&c) + } + + /// Relative href for a structured [`crate::path_types::ItemId`] in this scope’s garden. + pub(crate) fn garden_item_href(&self, c: &crate::path_types::ItemId) -> String { if let Some(tail) = c.tilde_tail().map(str::to_owned) { format!("{}/{}", self.garden_path_prefix, tail) } else if c.as_str().starts_with("http://") || c.as_str().starts_with("https://") { @@ -63,7 +68,7 @@ impl ThreadNav { let ext_prefix = format!("{}-", self.garden_path_prefix.trim_end_matches('~')); format!("{}/{}", ext_prefix, rest) } else { - format!("{}/{}", self.garden_path_prefix, canonicalize_item(item)) + format!("{}/{}", self.garden_path_prefix, canonicalize_item(c.as_str())) } } diff --git a/server/src/html/garden.rs b/server/src/html/garden.rs index e615dd356bcf634232d85610c0a26235ead125fd..2af9ac5bfdee1630883e6f8257883baeafcb5a44 100644 --- a/server/src/html/garden.rs +++ b/server/src/html/garden.rs @@ -10,7 +10,7 @@ use crate::{ api::optional_principal, canonical_path::canonicalize_item, events::ThreadCapability, - path_types::CanonicalItemUrl, + path_types::ItemId, reducer::{ContentState, ReducerState, ScopeId}, ranking::{connected_components_from_voted_pairs, ranked_items_subset}, scope_rank::{build_children_rankings, ChildrenRankings}, @@ -27,7 +27,7 @@ use super::{ /// Display path for an item: `~/…` or `-/…` form. fn item_display_path(item: &str) -> String { - CanonicalItemUrl::parse(item) + ItemId::parse(item) .map(|c| c.display_path()) .unwrap_or_else(|| canonicalize_item(item)) } @@ -159,7 +159,7 @@ pub async fn garden_index( let nav = ThreadNav::public(); let child_rankings = { let reduced = state.reduced.read().await; - build_children_rankings(reduced.public(), &CanonicalItemUrl::ontology_root()) + build_children_rankings(reduced.public(), &ItemId::ontology_root()) }; let page = layout( @@ -236,7 +236,7 @@ pub async fn external_garden_index( ) -> impl IntoResponse { let nav = ThreadNav::public(); let ext_path = ExternalOntologyPath::from_input(""); - let parent = CanonicalItemUrl::parse("https://.").unwrap(); + let parent = ItemId::parse("https://.").unwrap(); let child_rankings = { let reduced = state.reduced.read().await; build_children_rankings(reduced.public(), &parent) @@ -365,7 +365,7 @@ pub async fn room_external_garden_index( return room_not_found_page(&jar, &uri).into_response(); } let ext_path = ExternalOntologyPath::from_input(""); - let parent = CanonicalItemUrl::parse("https://.").unwrap(); + let parent = ItemId::parse("https://.").unwrap(); let child_rankings = build_children_rankings( content_for_garden_view(&reduced, &nav.scope()), &parent, @@ -509,13 +509,13 @@ struct ItemPageViewModel { fn build_sibling_rank( reduced: &crate::reducer::ReducerState, scope: &ScopeId, - item: &CanonicalItemUrl, + item: &ItemId, ) -> Option { let item = item.clone().normalized_storage(); let content = content_for_garden_view(reduced, scope); let group = &content.ranking_group; let parent = item.parent()?.normalized_storage(); - let siblings: Vec = content + let siblings: Vec = content .item_children .get(&parent) .map(|s| s.iter().cloned().collect()) @@ -574,7 +574,7 @@ fn build_rank_history( item: &str, ) -> Vec { let content = content_for_garden_view(reduced, scope); - let item_key = CanonicalItemUrl(item.to_string()); + let item_key = ItemId::parse(item).unwrap_or_else(|| ItemId::opaque(item.to_string())); let entries = match content.rank_history.get(&item_key) { None => return vec![], Some(e) => e, @@ -592,8 +592,8 @@ fn build_rank_history( if a_str == item || b_str == item { Some(crate::reducer::VoteData { ts: e.ts, - a: CanonicalItemUrl(a_str), - b: CanonicalItemUrl(b_str), + a: ItemId::parse(&a_str).unwrap_or_else(|| ItemId::opaque(a_str)), + b: ItemId::parse(&b_str).unwrap_or_else(|| ItemId::opaque(b_str)), ratio_left, ratio_right, body: explanation, principal: reduced.ingests_by_id.get(&e.post_id) @@ -629,8 +629,8 @@ fn build_item_page_view_model( item: &str, ) -> ItemPageViewModel { let content = content_for_garden_view(reduced, scope); - let item_key = CanonicalItemUrl::parse(item) - .unwrap_or_else(|| CanonicalItemUrl::parse("~/").unwrap()) + let item_key = ItemId::parse(item) + .unwrap_or_else(|| ItemId::parse("~/").unwrap()) .normalized_storage(); let item_has_parent = item_key.parent().is_some(); let child_rankings = build_children_rankings(content, &item_key); @@ -925,10 +925,10 @@ mod tests { .map(|r| r.item.as_str()) .collect(); assert_eq!(names, vec!["https://slug.social/~/topic/a", "https://slug.social/~/topic/b"]); - use crate::path_types::CanonicalItemUrl; + use crate::path_types::ItemId; assert!( - model.child_rankings.unranked_items.contains(&CanonicalItemUrl("https://slug.social/~/topic/kid1".to_string())) - || model.child_rankings.unranked_items.contains(&CanonicalItemUrl("https://slug.social/~/topic/kid2".to_string())) + model.child_rankings.unranked_items.contains(&ItemId::parse("https://slug.social/~/topic/kid1").unwrap()) + || model.child_rankings.unranked_items.contains(&ItemId::parse("https://slug.social/~/topic/kid2").unwrap()) ); } @@ -941,8 +941,8 @@ mod tests { "9ab12cd/my-room", "@00000000-0000-0000-0000-000000000000:test:local/test\n~/t1 {a}\n~/t2 {b}\n", ); - use crate::path_types::CanonicalItemUrl; - let root = CanonicalItemUrl::ontology_root(); + use crate::path_types::ItemId; + let root = ItemId::ontology_root(); let model = build_item_page_view_model( &reduced, &ScopeId::Room("9ab12cd/my-room".to_string()), @@ -971,8 +971,8 @@ mod tests { "@00000000-0000-0000-0000-000000000000:test:local/test\n\ ~/a {a}\n~/b {b}\n~/a 2:1 ~/b {because}\n", ); - use crate::path_types::CanonicalItemUrl; - let root = CanonicalItemUrl::ontology_root(); + use crate::path_types::ItemId; + let root = ItemId::ontology_root(); let model = build_item_page_view_model( &reduced, &ScopeId::Room("9ab12cd/my-room".to_string()), diff --git a/server/src/html/routing.rs b/server/src/html/routing.rs index 13e9ae0cbe32d454e34a7737edf8234d2a558502..df545a8195aa97484ef0012cff31d802ac67d9df 100644 --- a/server/src/html/routing.rs +++ b/server/src/html/routing.rs @@ -4,7 +4,7 @@ //! Prefer `RouteContext::item_href` / [`RouteContext::thread_url`] in new Maud over stitching //! `/r/…` vs `/~` manually. Call sites can migrate incrementally from passing `&ThreadNav`. -use crate::path_types::CanonicalItemUrl; +use crate::path_types::ItemId; use super::forum::ThreadNav; @@ -32,9 +32,9 @@ impl RouteContext { self.0 } - /// Relative path for a stored canonical item in this scope’s garden. - pub fn item_href(&self, item: &CanonicalItemUrl) -> String { - self.0.garden_item_url(item.as_str()) + /// Relative path for a stored item in this scope’s garden. + pub fn item_href(&self, item: &ItemId) -> String { + self.0.garden_item_href(item) } /// Same as [`Self::item_href`] but parses `item` first (raw DSL / user paste). diff --git a/server/src/path_types.rs b/server/src/path_types.rs index 361a9d446c9043cbdea5f060db2e8633c2fd9bf9..a6e0b90a1aaead0bc1f7e0f4c5296506edf3f7ac 100644 --- a/server/src/path_types.rs +++ b/server/src/path_types.rs @@ -1,5 +1,6 @@ -//! Re-exports — implementations live in `slug-types` (`paths` module). +//! Re-exports — implementations live in `slug-types` (`paths` / `item_id` modules). +pub use slug_types::ItemId; pub use slug_types::paths::{ - tilde_http_path_to_canonical, CanonicalItemUrl, RelativePath, TildeHttpPathTail, TildePath, + tilde_http_path_to_item_id, RelativePath, TildeHttpPathTail, TildePath, }; diff --git a/server/src/ranking.rs b/server/src/ranking.rs index b24e80ddd5f34b4e454bb81b541e8f54bd97a541..3710c9f64437f5bef3b2121905b6f3bcb7611047 100644 --- a/server/src/ranking.rs +++ b/server/src/ranking.rs @@ -1,11 +1,11 @@ use std::collections::HashMap; -use crate::path_types::CanonicalItemUrl; +use crate::path_types::ItemId; use crate::reducer::GroupState; #[derive(Debug, Clone)] pub struct RankedItem { - pub item: CanonicalItemUrl, + pub item: ItemId, pub score: f64, } @@ -239,7 +239,7 @@ pub fn group_summary_scores( group: &mut GroupState, max_iters: usize, tol: f64, -) -> HashMap { +) -> HashMap { ranked_items(group, max_iters, tol) .into_iter() .map(|r| (r.item, r.score)) @@ -256,11 +256,11 @@ mod tests { } fn vote(ts: i64, a: &str, b: &str, l: i32, r: i32) -> VoteData { - use crate::path_types::CanonicalItemUrl; + use crate::path_types::ItemId; VoteData { ts, - a: CanonicalItemUrl(a.to_string()), - b: CanonicalItemUrl(b.to_string()), + a: ItemId::parse(a).unwrap(), + b: ItemId::parse(b).unwrap(), ratio_left: l, ratio_right: r, body: "because".to_string(), diff --git a/server/src/reducer.rs b/server/src/reducer.rs index a8651994f2c062af25b6ad792a014dcc083296f9..cd11a5391de8e36b1c3bb6c9b8067ceba0878e31 100644 --- a/server/src/reducer.rs +++ b/server/src/reducer.rs @@ -5,7 +5,7 @@ use serde::{Deserialize, Serialize}; use crate::canonical_path::canonicalize_tag; use crate::dsl; use crate::events::{Event, Ingest, ThreadCapability}; -use crate::path_types::CanonicalItemUrl; +use crate::path_types::ItemId; #[derive(Debug, Clone, Hash, PartialEq, Eq, PartialOrd, Ord)] pub enum ScopeId { @@ -27,8 +27,8 @@ pub fn scope_from_room_wire(room: &str) -> ScopeId { #[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] pub struct VoteData { pub ts: i64, - pub a: CanonicalItemUrl, - pub b: CanonicalItemUrl, + pub a: ItemId, + pub b: ItemId, pub ratio_left: i32, pub ratio_right: i32, pub body: String, @@ -40,8 +40,8 @@ pub struct VoteData { #[derive(Debug, Clone)] pub struct GroupState { - pub item_to_idx: HashMap, - pub idx_to_item: Vec, + pub item_to_idx: HashMap, + pub idx_to_item: Vec, /// Aggregated directed edge weights: (src_idx, dst_idx) -> weight. pub edges: HashMap<(usize, usize), f64>, @@ -72,7 +72,7 @@ impl GroupState { } } - fn ensure_item(&mut self, item: &CanonicalItemUrl) -> usize { + fn ensure_item(&mut self, item: &ItemId) -> usize { if let Some(&idx) = self.item_to_idx.get(item) { return idx; } @@ -85,11 +85,11 @@ impl GroupState { /// Public test helper: insert an item into the group without a vote (for unit tests). pub fn ensure_item_pub(&mut self, item: &str) -> usize { - if let Some(canon) = CanonicalItemUrl::parse(item) { + if let Some(canon) = ItemId::parse(item) { self.ensure_item(&canon) } else { // Fallback: treat as raw canonical string - let canon = CanonicalItemUrl(item.to_string()); + let canon = ItemId::opaque(item.to_string()); self.ensure_item(&canon) } } @@ -103,8 +103,8 @@ impl GroupState { } pub fn apply_vote(&mut self, mut vote: VoteData) { - vote.a = CanonicalItemUrl::parse(vote.a.as_str()).unwrap_or(vote.a); - vote.b = CanonicalItemUrl::parse(vote.b.as_str()).unwrap_or(vote.b); + vote.a = ItemId::parse(vote.a.as_str()).unwrap_or_else(|| vote.a.clone()); + vote.b = ItemId::parse(vote.b.as_str()).unwrap_or_else(|| vote.b.clone()); vote.thread_tag = canonicalize_tag(&vote.thread_tag); if vote.ratio_left < 0 { vote.ratio_left = 0; @@ -210,18 +210,18 @@ impl Default for ForumThreadState { #[derive(Debug, Clone, Default)] pub struct ContentState { pub ranking_group: GroupState, - pub items: HashSet, - pub item_bodies: HashMap, + pub items: HashSet, + pub item_bodies: HashMap, /// Parent canonical URL -> direct children. - pub item_children: HashMap>, + pub item_children: HashMap>, /// Per-item vote history (most recent first). - pub item_votes: HashMap>, + pub item_votes: HashMap>, /// Per-item ingest references (most recent first). - pub item_snippets: HashMap>, + pub item_snippets: HashMap>, /// Item path -> threads that mention or vote on this item. - pub item_threads: HashMap>, + pub item_threads: HashMap>, /// Per-item rank history, oldest first. - pub rank_history: HashMap>, + pub rank_history: HashMap>, } #[derive(Debug, Clone)] @@ -351,7 +351,7 @@ impl ReducerState { /// For `a/b/c/d` this creates: `a/b/c→d`, `a/b→a/b/c`, `a→a/b`, `""→a`. /// Stops early when an intermediate is already registered (its ancestors must be too). /// @e2bdefa9-a6fa-4725-b0a2-c0b09d95bb20:claudecode:anthropic/claude-opus-4 - fn add_child_edge(content: &mut ContentState, item: &CanonicalItemUrl) { + fn add_child_edge(content: &mut ContentState, item: &ItemId) { let mut child = item.clone(); loop { let Some(parent) = child.parent() else { break }; @@ -366,16 +366,16 @@ impl ReducerState { } /// Resolve an item path as a first-class canonical path. - fn normalize_item(item: &str) -> Option { - CanonicalItemUrl::parse(item) + fn normalize_item(item: &str) -> Option { + ItemId::parse(item) } /// 1-indexed rank of `item` within its connected component in the parent scope. /// 0 if the item has no votes connecting it to siblings (unranked). fn scope_rank_of( group: &GroupState, - item: &CanonicalItemUrl, - item_children: &HashMap>, + item: &ItemId, + item_children: &HashMap>, ) -> usize { let scope = match item.parent() { Some(p) => p, @@ -421,7 +421,7 @@ impl ReducerState { /// 1-indexed position of `item` in the component-aware global flat list. /// Components sorted largest-first; items ranked within each component. /// 0 if the item is not in the ranking group. - fn global_rank_of(group: &GroupState, item: &CanonicalItemUrl) -> usize { + fn global_rank_of(group: &GroupState, item: &ItemId) -> usize { if !group.item_to_idx.contains_key(item) { return 0; } @@ -448,7 +448,7 @@ impl ReducerState { let doc = dsl::parse_full(&ing.raw).map_err(|_| ())?; let canonical_thread = canonicalize_tag(&ing.thread_tag); - let voted_items: Vec = doc + let voted_items: Vec = doc .statements .iter() .filter_map(|s| { @@ -467,7 +467,7 @@ impl ReducerState { let principal = ing.principal.clone(); let delegate = ing.delegate.clone(); - let before: HashMap = if !voted_items.is_empty() { + let before: HashMap = if !voted_items.is_empty() { crate::ranking::compute_group_ranking(&mut content.ranking_group, 10000, 1e-8); voted_items .iter() @@ -485,7 +485,7 @@ impl ReducerState { HashMap::new() }; - let mut ingest_items: HashSet = HashSet::new(); + let mut ingest_items: HashSet = HashSet::new(); for stmt in doc.statements { match stmt { diff --git a/server/src/scope_rank.rs b/server/src/scope_rank.rs index d656b2f0a0a3623ca6b234b746beaaca8ae6017d..4fdaae150c74ec5162c2accc06d4cc778ad79914 100644 --- a/server/src/scope_rank.rs +++ b/server/src/scope_rank.rs @@ -3,7 +3,7 @@ use std::collections::{HashMap, HashSet}; -use crate::path_types::CanonicalItemUrl; +use crate::path_types::ItemId; use crate::ranking::{connected_components_from_voted_pairs, ranked_items_subset, RankedItem}; use crate::reducer::ContentState; @@ -17,12 +17,12 @@ pub struct ScopedComponent { pub struct ChildrenRankings { pub component_rankings: Vec, /// Items in scope with no rank (no votes connecting them to others in this scope). - pub unranked_items: Vec, + pub unranked_items: Vec, } /// Resolve one scope spec (literal path) to direct children of that parent. No wildcards. -fn resolve_one_scope(content: &ContentState, spec: &str) -> HashSet { - let Some(parent) = CanonicalItemUrl::parse(spec.trim()) else { +fn resolve_one_scope(content: &ContentState, spec: &str) -> HashSet { + let Some(parent) = ItemId::parse(spec.trim()) else { return HashSet::new(); }; content @@ -36,12 +36,12 @@ fn resolve_one_scope(content: &ContentState, spec: &str) -> HashSet Vec { +pub fn resolve_scope(content: &ContentState, specs: &[String]) -> Vec { let mut set = HashSet::new(); for spec in specs { set.extend(resolve_one_scope(content, spec)); } - let mut out: Vec = set.into_iter().collect(); + let mut out: Vec = set.into_iter().collect(); out.sort(); out } @@ -49,18 +49,18 @@ pub fn resolve_scope(content: &ContentState, specs: &[String]) -> Vec Vec { +pub fn resolve_scope_recursive(content: &ContentState, specs: &[String], depth: usize) -> Vec { if depth == 0 { return vec![]; } - let mut visited: HashSet = HashSet::new(); - let mut frontier: Vec = specs + let mut visited: HashSet = HashSet::new(); + let mut frontier: Vec = specs .iter() - .filter_map(|s| CanonicalItemUrl::parse(s)) + .filter_map(|s| ItemId::parse(s)) .collect(); for _level in 0..depth { - let mut next_frontier: Vec = Vec::new(); + let mut next_frontier: Vec = Vec::new(); for parent in &frontier { if let Some(children) = content.item_children.get(parent) { for child in children { @@ -76,16 +76,16 @@ pub fn resolve_scope_recursive(content: &ContentState, specs: &[String], depth: frontier = next_frontier; } - let mut out: Vec = visited.into_iter().collect(); + let mut out: Vec = visited.into_iter().collect(); out.sort(); out } /// Build connected-component rankings for an explicit set of item paths. /// Use this when scope comes from multiple parents (resolve_scope). -pub fn build_rankings_for_item_set(content: &ContentState, items_in_scope: &[CanonicalItemUrl]) -> ChildrenRankings { +pub fn build_rankings_for_item_set(content: &ContentState, items_in_scope: &[ItemId]) -> ChildrenRankings { let group = &content.ranking_group; - let mut items_in_scope: Vec = items_in_scope.to_vec(); + let mut items_in_scope: Vec = items_in_scope.to_vec(); items_in_scope.sort(); let scoped_idxs: Vec = items_in_scope @@ -127,7 +127,7 @@ pub fn build_rankings_for_item_set(content: &ContentState, items_in_scope: &[Can }) .collect(); - let mut unranked_items: Vec = isolate_local_idxs + let mut unranked_items: Vec = isolate_local_idxs .into_iter() .filter_map(|li| local_to_global.get(li).copied()) .filter_map(|idx| group.idx_to_item.get(idx).cloned()) @@ -148,9 +148,9 @@ pub fn build_rankings_for_item_set(content: &ContentState, items_in_scope: &[Can /// Build connected-component rankings for direct children of parent_scope. /// Matches the HTML garden view: multiple components, isolates, no-vote items. -pub fn build_children_rankings(content: &ContentState, parent: &CanonicalItemUrl) -> ChildrenRankings { +pub fn build_children_rankings(content: &ContentState, parent: &ItemId) -> ChildrenRankings { let parent = parent.clone().normalized_storage(); - let items: Vec = content + let items: Vec = content .item_children .get(&parent) .map(|s| s.iter().cloned().collect()) @@ -164,10 +164,13 @@ mod tests { use std::collections::{HashMap, HashSet}; fn content_with_children(edges: &[(&str, &[&str])]) -> ContentState { - let mut item_children: HashMap> = HashMap::new(); + let mut item_children: HashMap> = HashMap::new(); for (parent, children) in edges { - let parent = CanonicalItemUrl((*parent).to_string()); - let set: HashSet = children.iter().map(|s| CanonicalItemUrl((*s).to_string())).collect(); + let parent = ItemId::parse(parent).unwrap(); + let set: HashSet = children + .iter() + .map(|s| ItemId::parse(s).unwrap()) + .collect(); item_children.insert(parent, set); } ContentState { @@ -189,8 +192,8 @@ mod tests { ]); let out = resolve_one_scope(&content, "models"); assert_eq!(out.len(), 2); - assert!(out.contains(&CanonicalItemUrl("https://slug.social/models/x".to_string()))); - assert!(out.contains(&CanonicalItemUrl("https://slug.social/models/y".to_string()))); + assert!(out.contains(&ItemId::parse("https://slug.social/models/x").unwrap())); + assert!(out.contains(&ItemId::parse("https://slug.social/models/y").unwrap())); } #[test] @@ -201,9 +204,9 @@ mod tests { ]); let out = resolve_scope(&content, &["a".into(), "b".into()]); assert_eq!(out.len(), 4); - assert!(out.contains(&CanonicalItemUrl("https://slug.social/a/1".to_string()))); - assert!(out.contains(&CanonicalItemUrl("https://slug.social/a/2".to_string()))); - assert!(out.contains(&CanonicalItemUrl("https://slug.social/b/1".to_string()))); - assert!(out.contains(&CanonicalItemUrl("https://slug.social/b/2".to_string()))); + assert!(out.contains(&ItemId::parse("https://slug.social/a/1").unwrap())); + assert!(out.contains(&ItemId::parse("https://slug.social/a/2").unwrap())); + assert!(out.contains(&ItemId::parse("https://slug.social/b/1").unwrap())); + assert!(out.contains(&ItemId::parse("https://slug.social/b/2").unwrap())); } } diff --git a/server/tests/basic.rs b/server/tests/basic.rs index 17db86894f18d6b542d2943f4cedb0572b247171..c8efea2976e20b443d78a5d4c3b244bc4f2fd6da 100644 --- a/server/tests/basic.rs +++ b/server/tests/basic.rs @@ -7,8 +7,15 @@ use slugsocial_server::{ }; +use slugsocial_server::path_types::ItemId; + use tempfile::TempDir; +#[inline] +fn item_id(s: &str) -> ItemId { + ItemId::parse(s).unwrap() +} + fn ingest_event(ts: i64, raw: &str) -> Event { Event::Ingest(Ingest { ts, @@ -47,7 +54,7 @@ fn reducer_external_namespace_ranking() { let g = &content.ranking_group; assert_eq!(g.idx_to_item.len(), 2); - let parent = slugsocial_server::path_types::CanonicalItemUrl::parse("https://github.com/iss").unwrap(); + let parent = slugsocial_server::path_types::ItemId::parse("https://github.com/iss").unwrap(); let children = slugsocial_server::scope_rank::build_children_rankings(content, &parent); assert_eq!(children.component_rankings.len(), 1); let names: Vec<&str> = children.component_rankings[0] @@ -77,9 +84,9 @@ fn reducer_and_ranking_linear_chain() { let mut group = state.public().ranking_group.clone(); let ranked = ranked_items(&mut group, 20000, 1e-9); assert_eq!(ranked.len(), 3); - assert_eq!(ranked[0].item, "https://slug.social/~/t/a"); - assert_eq!(ranked[1].item, "https://slug.social/~/t/b"); - assert_eq!(ranked[2].item, "https://slug.social/~/t/c"); + assert_eq!(ranked[0].item.as_str(), "https://slug.social/~/t/a"); + assert_eq!(ranked[1].item.as_str(), "https://slug.social/~/t/b"); + assert_eq!(ranked[2].item.as_str(), "https://slug.social/~/t/c"); } #[test] @@ -112,15 +119,15 @@ fn reducer_handles_item_and_body_from_ingest() { )); let content = state.public(); - assert!(content.items.contains("https://slug.social/~/t/test-item")); + assert!(content.items.contains(&item_id("https://slug.social/~/t/test-item"))); assert_eq!( - content.item_bodies.get("https://slug.social/~/t/test-item"), + content.item_bodies.get(&item_id("https://slug.social/~/t/test-item")), Some(&"Description here".to_string()) ); assert!(content .item_children - .get("https://slug.social/~/t") - .map(|c| c.contains("https://slug.social/~/t/test-item")) + .get(&item_id("https://slug.social/~/t")) + .map(|c| c.contains(&item_id("https://slug.social/~/t/test-item"))) .unwrap_or(false)); } @@ -136,12 +143,12 @@ fn reducer_indexes_item_threads_and_vote_thread() { state.apply_event(Event::Ingest(ev)); let content = state.public(); - let threads_for_insertion = content.item_threads.get("https://slug.social/~/sorts/insertion").unwrap(); + let threads_for_insertion = content.item_threads.get(&item_id("https://slug.social/~/sorts/insertion")).unwrap(); assert!(threads_for_insertion.contains("sorting-hat")); - let threads_for_mergesort = content.item_threads.get("https://slug.social/~/sorts/mergesort").unwrap(); + let threads_for_mergesort = content.item_threads.get(&item_id("https://slug.social/~/sorts/mergesort")).unwrap(); assert!(threads_for_mergesort.contains("sorting-hat")); - let vote = content.item_votes.get("https://slug.social/~/sorts/insertion").unwrap().front().unwrap(); + let vote = content.item_votes.get(&item_id("https://slug.social/~/sorts/insertion")).unwrap().front().unwrap(); assert_eq!(vote.thread_tag, "sorting-hat"); } @@ -155,8 +162,8 @@ fn reducer_aggregates_multiple_votes() { } let group = &state.public().ranking_group; - let a_idx = group.item_to_idx["https://slug.social/~/t/a"]; - let b_idx = group.item_to_idx["https://slug.social/~/t/b"]; + let a_idx = group.item_to_idx[&item_id("https://slug.social/~/t/a")]; + let b_idx = group.item_to_idx[&item_id("https://slug.social/~/t/b")]; // Should have accumulated edge weights in both directions. assert!(group.edges.contains_key(&(a_idx, b_idx))); @@ -232,7 +239,7 @@ fn ranking_dominant_item_wins() { let mut group = state.public().ranking_group.clone(); let ranked = ranked_items(&mut group, 20000, 1e-9); - assert_eq!(ranked[0].item, "https://slug.social/~/t/champion"); + assert_eq!(ranked[0].item.as_str(), "https://slug.social/~/t/champion"); assert!(ranked[0].score > ranked[1].score); } @@ -379,7 +386,7 @@ async fn full_workflow_reducer_and_ranking() { let ranked = ranked_items(&mut group, 20000, 1e-9); assert_eq!(ranked.len(), 2); - assert_eq!(ranked[0].item, "https://slug.social/~/langs/rust"); // Should win + assert_eq!(ranked[0].item.as_str(), "https://slug.social/~/langs/rust"); // Should win assert!(ranked[0].score > ranked[1].score); } @@ -395,27 +402,27 @@ fn reducer_materializes_ancestor_path_segments() { // The intermediate path "https://slug.social/~/ai-models/anthropic" should appear as a child of "https://slug.social/~/ai-models". let content = state.public(); - let ai_models_children = content.item_children.get("https://slug.social/~/ai-models").expect("ai-models should have children"); + let ai_models_children = content.item_children.get(&item_id("https://slug.social/~/ai-models")).expect("ai-models should have children"); assert!( - ai_models_children.contains("https://slug.social/~/ai-models/anthropic"), + ai_models_children.contains(&item_id("https://slug.social/~/ai-models/anthropic")), "ai-models/anthropic should be a child of ai-models" ); // The leaf items should still be children of "https://slug.social/~/ai-models/anthropic". - let anthropic_children = content.item_children.get("https://slug.social/~/ai-models/anthropic").expect("ai-models/anthropic should have children"); - assert!(anthropic_children.contains("https://slug.social/~/ai-models/anthropic/claude-opus")); - assert!(anthropic_children.contains("https://slug.social/~/ai-models/anthropic/claude-sonnet")); + let anthropic_children = content.item_children.get(&item_id("https://slug.social/~/ai-models/anthropic")).expect("ai-models/anthropic should have children"); + assert!(anthropic_children.contains(&item_id("https://slug.social/~/ai-models/anthropic/claude-opus"))); + assert!(anthropic_children.contains(&item_id("https://slug.social/~/ai-models/anthropic/claude-sonnet"))); // Root should contain "https://slug.social/~/ai-models". - let root_children = content.item_children.get("https://slug.social/~").expect("root should have children"); - assert!(root_children.contains("https://slug.social/~/ai-models")); + let root_children = content.item_children.get(&item_id("https://slug.social/~")).expect("root should have children"); + assert!(root_children.contains(&item_id("https://slug.social/~/ai-models"))); // The phantom intermediates should NOT be in the items set (they weren't explicitly created). - assert!(!content.items.contains("https://slug.social/~/ai-models")); - assert!(!content.items.contains("https://slug.social/~/ai-models/anthropic")); + assert!(!content.items.contains(&item_id("https://slug.social/~/ai-models"))); + assert!(!content.items.contains(&item_id("https://slug.social/~/ai-models/anthropic"))); // But the leaf items should be. - assert!(content.items.contains("https://slug.social/~/ai-models/anthropic/claude-opus")); - assert!(content.items.contains("https://slug.social/~/ai-models/anthropic/claude-sonnet")); + assert!(content.items.contains(&item_id("https://slug.social/~/ai-models/anthropic/claude-opus"))); + assert!(content.items.contains(&item_id("https://slug.social/~/ai-models/anthropic/claude-sonnet"))); } #[test] @@ -469,9 +476,9 @@ fn ranking_repeated_votes_normalized() { // Same winner regardless of how many times voted. assert_eq!(ranked_once[0].item, ranked_many[0].item); - assert_eq!(ranked_once[0].item, "https://slug.social/~/norm/a"); + assert_eq!(ranked_once[0].item.as_str(), "https://slug.social/~/norm/a"); assert_eq!(ranked_once[1].item, ranked_many[1].item); - assert_eq!(ranked_once[1].item, "https://slug.social/~/norm/b"); + assert_eq!(ranked_once[1].item.as_str(), "https://slug.social/~/norm/b"); // Scores should be identical (normalization makes repeated votes idempotent). let eps = 1e-6; @@ -506,8 +513,8 @@ fn reducer_zero_zero_vote_ratio_normalizes_to_one_one() { "~/t/a {a}\n~/t/b {b}\n~/t/a 0:0 ~/t/b {zero}\n", )); let group = &state.public().ranking_group; - let a_idx = group.item_to_idx["https://slug.social/~/t/a"]; - let b_idx = group.item_to_idx["https://slug.social/~/t/b"]; + let a_idx = group.item_to_idx[&item_id("https://slug.social/~/t/a")]; + let b_idx = group.item_to_idx[&item_id("https://slug.social/~/t/b")]; // 0:0 should normalize to 1:1 — both directions should have weight assert!(group.edges.contains_key(&(a_idx, b_idx))); assert!(group.edges.contains_key(&(b_idx, a_idx))); @@ -523,8 +530,8 @@ fn reducer_negative_ratio_clamped_to_zero() { let mut group = GroupState::new(); group.apply_vote(slugsocial_server::reducer::VoteData { ts: 1, - a: slugsocial_server::path_types::CanonicalItemUrl("https://slug.social/~/t/a".to_string()), - b: slugsocial_server::path_types::CanonicalItemUrl("https://slug.social/~/t/b".to_string()), + a: slugsocial_server::path_types::ItemId::parse("https://slug.social/~/t/a").unwrap(), + b: slugsocial_server::path_types::ItemId::parse("https://slug.social/~/t/b").unwrap(), ratio_left: -5, ratio_right: -3, body: "negative".to_string(), @@ -534,8 +541,8 @@ fn reducer_negative_ratio_clamped_to_zero() { }); assert_eq!(group.idx_to_item.len(), 2); // Both edges should exist (negatives clamped to 0, then 0:0 -> 1:1) - let a_idx = group.item_to_idx["https://slug.social/~/t/a"]; - let b_idx = group.item_to_idx["https://slug.social/~/t/b"]; + let a_idx = group.item_to_idx[&item_id("https://slug.social/~/t/a")]; + let b_idx = group.item_to_idx[&item_id("https://slug.social/~/t/b")]; assert!(group.edges.contains_key(&(a_idx, b_idx))); assert!(group.edges.contains_key(&(b_idx, a_idx))); } @@ -553,24 +560,24 @@ fn reducer_deep_path_ancestor_materialization_four_levels() { let content = state.public(); let tilde_scope = content .item_children - .get("https://slug.social/~") + .get(&item_id("https://slug.social/~")) .expect("~/ scope should have children"); - assert!(tilde_scope.contains("https://slug.social/~/a")); + assert!(tilde_scope.contains(&item_id("https://slug.social/~/a"))); - let a_children = content.item_children.get("https://slug.social/~/a").expect("a should have children"); - assert!(a_children.contains("https://slug.social/~/a/b")); + let a_children = content.item_children.get(&item_id("https://slug.social/~/a")).expect("a should have children"); + assert!(a_children.contains(&item_id("https://slug.social/~/a/b"))); - let ab_children = content.item_children.get("https://slug.social/~/a/b").expect("a/b should have children"); - assert!(ab_children.contains("https://slug.social/~/a/b/c")); + let ab_children = content.item_children.get(&item_id("https://slug.social/~/a/b")).expect("a/b should have children"); + assert!(ab_children.contains(&item_id("https://slug.social/~/a/b/c"))); - let abc_children = content.item_children.get("https://slug.social/~/a/b/c").expect("a/b/c should have children"); - assert!(abc_children.contains("https://slug.social/~/a/b/c/d")); + let abc_children = content.item_children.get(&item_id("https://slug.social/~/a/b/c")).expect("a/b/c should have children"); + assert!(abc_children.contains(&item_id("https://slug.social/~/a/b/c/d"))); // Only the leaf should be in items set - assert!(content.items.contains("https://slug.social/~/a/b/c/d")); - assert!(!content.items.contains("https://slug.social/~/a")); - assert!(!content.items.contains("https://slug.social/~/a/b")); - assert!(!content.items.contains("https://slug.social/~/a/b/c")); + assert!(content.items.contains(&item_id("https://slug.social/~/a/b/c/d"))); + assert!(!content.items.contains(&item_id("https://slug.social/~/a"))); + assert!(!content.items.contains(&item_id("https://slug.social/~/a/b"))); + assert!(!content.items.contains(&item_id("https://slug.social/~/a/b/c"))); } // ============================================================================ @@ -610,7 +617,7 @@ fn ranking_convergence_tolerance_triggers_early_exit() { // Very tight tolerance but huge max_iters — should still converge fast let ranked = ranked_items(&mut group, 1_000_000, 1e-15); assert_eq!(ranked.len(), 2); - assert_eq!(ranked[0].item, "https://slug.social/~/t/a"); + assert_eq!(ranked[0].item.as_str(), "https://slug.social/~/t/a"); } // ============================================================================ @@ -624,14 +631,14 @@ fn test_item_body_overwrite() { 1, "~/t/x {first}\n", )); - assert_eq!(state.public().item_bodies.get("https://slug.social/~/t/x"), Some(&"first".to_string())); + assert_eq!(state.public().item_bodies.get(&item_id("https://slug.social/~/t/x")), Some(&"first".to_string())); state.apply_event(ingest_event( 2, "~/t/x {second}\n", )); assert_eq!( - state.public().item_bodies.get("https://slug.social/~/t/x"), + state.public().item_bodies.get(&item_id("https://slug.social/~/t/x")), Some(&"second".to_string()), "last writer should win for item bodies" ); @@ -645,9 +652,9 @@ fn test_empty_body_not_stored() { "~/t/blank { }\n", )); let content = state.public(); - assert!(content.items.contains("https://slug.social/~/t/blank"), "item should exist"); + assert!(content.items.contains(&item_id("https://slug.social/~/t/blank")), "item should exist"); assert!( - !content.item_bodies.contains_key("https://slug.social/~/t/blank"), + !content.item_bodies.contains_key(&item_id("https://slug.social/~/t/blank")), "whitespace-only body should not be stored" ); } @@ -664,7 +671,7 @@ fn test_duplicate_items_across_ingests() { "~/t/dup {second}\n", )); let content = state.public(); - let count = content.items.iter().filter(|i| *i == "https://slug.social/~/t/dup").count(); + let count = content.items.iter().filter(|i| i.as_str() == "https://slug.social/~/t/dup").count(); assert_eq!(count, 1, "items set should deduplicate across ingests"); } @@ -736,7 +743,7 @@ fn test_thread_id_is_used_for_votes_and_indexes() { let content = state.public(); let vote = content .item_votes - .get("https://slug.social/~/t/a") + .get(&item_id("https://slug.social/~/t/a")) .unwrap() .front() .unwrap(); @@ -745,7 +752,7 @@ fn test_thread_id_is_used_for_votes_and_indexes() { assert!(state .public() .item_threads - .get("https://slug.social/~/t/a") + .get(&item_id("https://slug.social/~/t/a")) .is_some_and(|threads| threads.contains("first"))); } @@ -757,11 +764,11 @@ fn test_rank_history_created_for_voted_items() { "~/t/a {a}\n~/t/b {b}\n~/t/a 3:1 ~/t/b {reason}\n", )); assert!( - state.public().rank_history.contains_key("https://slug.social/~/t/a"), + state.public().rank_history.contains_key(&item_id("https://slug.social/~/t/a")), "rank_history should have entry for voted item a" ); assert!( - state.public().rank_history.contains_key("https://slug.social/~/t/b"), + state.public().rank_history.contains_key(&item_id("https://slug.social/~/t/b")), "rank_history should have entry for voted item b" ); } @@ -774,7 +781,7 @@ fn test_rank_history_not_created_for_unvoted_items() { "~/t/c {just a definition}\n", )); assert!( - !state.public().rank_history.contains_key("https://slug.social/~/t/c"), + !state.public().rank_history.contains_key(&item_id("https://slug.social/~/t/c")), "rank_history should NOT have entry for item with no votes" ); } @@ -786,7 +793,7 @@ fn test_rank_history_first_entry_delta_zero() { 1, "~/t/a {a}\n~/t/b {b}\n~/t/a 3:1 ~/t/b {reason}\n", )); - let history_a = state.public().rank_history.get("https://slug.social/~/t/a").unwrap(); + let history_a = state.public().rank_history.get(&item_id("https://slug.social/~/t/a")).unwrap(); assert_eq!(history_a.len(), 1); assert_eq!( history_a[0].scope_rank_delta, 0, diff --git a/types/src/item_id.rs b/types/src/item_id.rs new file mode 100644 index 0000000000000000000000000000000000000000..5fe9fad39c0d58e546c0a0162eba4348d64e0865 --- /dev/null +++ b/types/src/item_id.rs @@ -0,0 +1,225 @@ +//! Structural item identity for the reducer graph and ranking (vs presentation-only strings). + +use std::cmp::Ordering; +use std::fmt; + +use serde::{Deserialize, Deserializer, Serialize, Serializer}; + +use crate::item_wire::{ + canonicalize_item, external_display_dash_prefix, normalize_slug_ontology_storage_url, + SLUG_TILDE_ONTOLOGY_ROOT, +}; + +/// Structural key for items in [`slug_types`] and the server reducer. +/// +/// Wire / JSON uses the same single string as the former canonical item URL (via serde). +#[derive(Debug, Clone, Hash, PartialEq, Eq)] +pub enum ItemId { + /// Tilde ontology root (`~/`); storage [`SLUG_TILDE_ONTOLOGY_ROOT`]. + Root, + /// `https://slug.social/~/…` (non-root; normalized trailing path). + Local(String), + /// Normalized `http(s)://…` storage form, including non-`~/` paths on `slug.social`. + Web(String), + /// Raw key material that did not round-trip through [`Self::parse`] (historical edge case). + Opaque(String), +} + +impl ItemId { + pub fn parse(input: &str) -> Option { + let c = normalize_slug_ontology_storage_url(&canonicalize_item(input)); + if c.is_empty() { + return None; + } + if c == SLUG_TILDE_ONTOLOGY_ROOT { + return Some(Self::Root); + } + if c.starts_with("https://slug.social/~/") && c != SLUG_TILDE_ONTOLOGY_ROOT { + return Some(Self::Local(c)); + } + Some(Self::Web(c)) + } + + /// Same as the old `ensure_item` fallback: use `s` verbatim as the map key. + pub fn opaque(raw: String) -> Self { + Self::Opaque(raw) + } + + pub fn ontology_root() -> Self { + Self::Root + } + + /// Collapses legacy slug ontology root spellings so [`std::collections::HashMap`] keys match the graph. + pub fn normalized_storage(self) -> Self { + Self::parse(self.as_str()).unwrap_or(self) + } + + pub fn as_str(&self) -> &str { + match self { + ItemId::Root => SLUG_TILDE_ONTOLOGY_ROOT, + ItemId::Local(s) | ItemId::Web(s) | ItemId::Opaque(s) => s, + } + } + + pub fn to_storage_string(&self) -> String { + self.as_str().to_string() + } + + pub fn tilde_tail(&self) -> Option<&str> { + match self { + ItemId::Root => Some(""), + ItemId::Local(s) => s.strip_prefix("https://slug.social/~/"), + ItemId::Web(s) | ItemId::Opaque(s) => { + if let Some(tail) = s.strip_prefix("https://slug.social/~/") { + return Some(tail); + } + if s == SLUG_TILDE_ONTOLOGY_ROOT || s == "https://slug.social/~/" { + return Some(""); + } + None + } + } + } + + /// HTTP garden tail after `~/` (empty at ontology root), or `None` if not under tilde ontology. + pub fn tilde_http_tail(&self) -> Option { + self.tilde_tail().map(str::to_owned) + } + + pub fn last_segment(&self) -> &str { + let s = self.as_str(); + s.rsplit('/').find(|x| !x.is_empty()).unwrap_or(s) + } + + pub fn parent(&self) -> Option { + match self { + ItemId::Root => None, + ItemId::Local(s) => { + if s == SLUG_TILDE_ONTOLOGY_ROOT || s == "https://slug.social/~/" { + return None; + } + let last_slash = s.rfind('/')?; + let parent_str = &s[..last_slash]; + if parent_str.is_empty() { + None + } else { + Self::parse(parent_str) + } + } + ItemId::Web(s) | ItemId::Opaque(s) => { + if s == SLUG_TILDE_ONTOLOGY_ROOT || s == "https://slug.social/~/" { + return None; + } + if let Some(rest) = s.strip_prefix("https://slug.social/~/") { + if rest.is_empty() { + return None; + } + let last_slash = s.rfind('/')?; + let parent_str = &s[..last_slash]; + Self::parse(parent_str) + } else if s.starts_with("https://") { + let rest = s.strip_prefix("https://").unwrap(); + Self::parent_http_url("https://", rest) + } else if s.starts_with("http://") { + let rest = s.strip_prefix("http://").unwrap(); + Self::parent_http_url("http://", rest) + } else { + None + } + } + } + } + + fn parent_http_url(scheme: &'static str, rest: &str) -> Option { + let (host, path) = rest.split_once('/').map_or((rest, ""), |(h, p)| (h, p)); + let host = host.trim(); + let path = path.trim_end_matches('/'); + if path.is_empty() { + return None; + } + let parent_path = path.rsplit_once('/').map(|(p, _)| p).unwrap_or(""); + if parent_path.is_empty() { + Self::parse(&format!("{scheme}{}", host)) + } else { + Self::parse(&format!("{scheme}{}/{}", host, parent_path)) + } + } + + /// `-/` representation for external `https://…` items, `~/…` for slug ontology, else unchanged. + pub fn display_path(&self) -> String { + if let Some(tail) = self.tilde_tail() { + if tail.is_empty() { + return "~/".to_string(); + } + return format!("~/{}", tail); + } + let s = self.as_str(); + if let Some(tail) = s.strip_prefix("https://") { + if tail.starts_with("slug.social") { + return s.to_string(); + } + return external_display_dash_prefix(tail); + } + if let Some(tail) = s.strip_prefix("http://") { + if tail.starts_with("slug.social") { + return s.to_string(); + } + return external_display_dash_prefix(tail); + } + s.to_string() + } + + pub fn tilde_segments(&self) -> Vec<&str> { + match self.tilde_tail() { + Some(tail) if !tail.is_empty() => std::iter::once("~") + .chain(tail.split('/').filter(|s| !s.is_empty())) + .collect(), + Some(_) => vec!["~"], + None => vec![], + } + } + + /// Normalized URL string for HTTP fetch boundaries (external identities). + pub fn to_wire_url(&self) -> String { + self.to_storage_string() + } +} + +impl PartialOrd for ItemId { + fn partial_cmp(&self, other: &Self) -> Option { + Some(self.cmp(other)) + } +} + +impl Ord for ItemId { + fn cmp(&self, other: &Self) -> Ordering { + self.as_str().cmp(other.as_str()) + } +} + +impl fmt::Display for ItemId { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.as_str().fmt(f) + } +} + +impl Serialize for ItemId { + fn serialize(&self, serializer: S) -> Result + where + S: Serializer, + { + self.as_str().serialize(serializer) + } +} + +impl<'de> Deserialize<'de> for ItemId { + fn deserialize(deserializer: D) -> Result + where + D: Deserializer<'de>, + { + let s = String::deserialize(deserializer)?; + ItemId::parse(&s).ok_or_else(|| { + serde::de::Error::custom(format!("invalid item id: {s:?}")) + }) + } +} diff --git a/types/src/item_wire.rs b/types/src/item_wire.rs new file mode 100644 index 0000000000000000000000000000000000000000..20e3e7df77a74713c146c7e7222a809173723c7d --- /dev/null +++ b/types/src/item_wire.rs @@ -0,0 +1,184 @@ +//! Wire normalization for item identity strings (`canonicalize_item`, path segments). +//! Split from `paths` so [`crate::item_id::ItemId`] can depend on this without import cycles. + +use crate::url_normalize::{host_preserves_dash_path_case, normalize_http_identity_url}; + +/// Canonical absolute URL for the tilde ontology **root** (`~/` in UI). +pub const SLUG_TILDE_ONTOLOGY_ROOT: &str = "https://slug.social/~"; + +/// Collapse legacy or parser variants of the ontology root to [`SLUG_TILDE_ONTOLOGY_ROOT`]. +pub fn normalize_slug_ontology_storage_url(s: &str) -> String { + if s == "https://slug.social/~/" { + SLUG_TILDE_ONTOLOGY_ROOT.to_string() + } else { + s.to_string() + } +} + +fn finalize_external_identity_url(s: String) -> String { + if s.starts_with("https://slug.social/") { + return s; + } + let normalized = normalize_http_identity_url(&s).unwrap_or_else(|| s.clone()); + strip_redundant_root_slash(&normalized).unwrap_or(normalized) +} + +/// `url::Url` serializes bare hosts with a `/` path; we keep host-only items slash-free for stable +/// keys matching the pre-normalizer spellings. +fn strip_redundant_root_slash(s: &str) -> Option { + let u = url::Url::parse(s).ok()?; + if u.path() == "/" && u.query().is_none() && u.fragment().is_none() { + let scheme = u.scheme(); + let host = u.host_str()?; + return Some(match u.port() { + Some(p) => format!("{scheme}://{host}:{p}"), + None => format!("{scheme}://{host}"), + }); + } + None +} + +/// Ontology item reference → canonical absolute URL on the slug host. +pub fn canonicalize_item(input: &str) -> String { + let s = input.trim(); + if s.is_empty() { + return String::new(); + } + + if let Some(rest) = s.strip_prefix("-/") { + let (host, tail) = rest + .split_once('/') + .map_or((rest, ""), |(h, t)| (h, t)); + let host = host.trim().to_lowercase(); + if host.is_empty() { + return String::new(); + } + let preserve_case = host_preserves_dash_path_case(&host); + return if tail.is_empty() { + finalize_external_identity_url(format!("https://{}", host)) + } else { + let path = tail + .trim_start_matches('/') + .trim_end_matches('/') + .split('/') + .filter_map(|seg| { + let t = seg.trim(); + if t.is_empty() { + None + } else if preserve_case { + Some(t.to_string()) + } else { + Some(t.to_lowercase()) + } + }) + .collect::>() + .join("/"); + finalize_external_identity_url(format!("https://{}/{}", host, path)) + }; + } + + if let Some(rest) = s.strip_prefix("https://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let host = host.trim().to_lowercase(); + return finalize_external_identity_url(if tail.is_empty() { + format!("https://{}", host) + } else { + format!("https://{}/{}", host, tail) + }); + } + if let Some(rest) = s.strip_prefix("http://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let host = host.trim().to_lowercase(); + return finalize_external_identity_url(if tail.is_empty() { + format!("http://{}", host) + } else { + format!("http://{}/{}", host, tail) + }); + } + + let is_tilde = s.starts_with("~/"); + let rest = s.strip_prefix("~/").or_else(|| s.strip_prefix("/")).unwrap_or(s); + + let tail = rest + .split('/') + .filter_map(|seg| { + let t = seg.trim(); + if t.is_empty() { + None + } else { + Some(t.to_lowercase()) + } + }) + .collect::>() + .join("/"); + + if is_tilde { + if tail.is_empty() { + return SLUG_TILDE_ONTOLOGY_ROOT.to_string(); + } + format!("https://slug.social/~/{}", tail) + } else if tail.is_empty() { + "https://slug.social".to_string() + } else { + format!("https://slug.social/{}", tail) + } +} + +pub fn item_path_segments(input: &str) -> Vec { + let canonical = canonicalize_item(input); + if canonical.is_empty() { + return vec![]; + } + + if let Some(rest) = canonical.strip_prefix("https://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let mut out = vec![format!("https://{}", host)]; + out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); + return out; + } + if let Some(rest) = canonical.strip_prefix("http://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let mut out = vec![format!("http://{}", host)]; + out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); + return out; + } + + canonical + .split('/') + .filter(|s| !s.is_empty()) + .map(|s| s.to_string()) + .collect() +} + +pub fn item_parent_path(input: &str) -> Option { + let segs = item_path_segments(input); + if segs.len() <= 1 { + return None; + } + Some(segs[..segs.len() - 1].join("/")) +} + +pub(crate) fn external_display_dash_prefix(host_and_path: &str) -> String { + let (host, path) = host_and_path + .split_once('/') + .map_or((host_and_path, ""), |(h, p)| (h, p)); + let host = host.trim().to_lowercase(); + let path = path + .trim_end_matches('/') + .split('/') + .filter_map(|seg| { + let t = seg.trim(); + if t.is_empty() { + None + } else { + Some(t.to_lowercase()) + } + }) + .collect::>() + .join("/"); + if path.is_empty() { + format!("-/{}", host) + } else { + format!("-/{}", format!("{}/{}", host, path)) + } +} diff --git a/types/src/lib.rs b/types/src/lib.rs index ceaa574cd32b7e7eb397c9a3191fcbc037b8c606..05bcfcfbf48f1dc7a4c05e32055bd12eb5cfd623 100644 --- a/types/src/lib.rs +++ b/types/src/lib.rs @@ -1,14 +1,20 @@ use serde::{Deserialize, Serialize}; pub mod url_normalize; +pub mod item_wire; +pub mod item_id; pub mod paths; pub mod timeago; +pub use item_id::ItemId; +pub use item_wire::{ + canonicalize_item, item_parent_path, item_path_segments, normalize_slug_ontology_storage_url, + SLUG_TILDE_ONTOLOGY_ROOT, +}; pub use paths::{ - canonicalize_item, canonicalize_tag, item_parent_path, item_path_segments, normalize_slug_ontology_storage_url, - CanonicalItemUrl, ForumThreadUrl, GardenItemUrl, RelativePath, SLUG_TILDE_ONTOLOGY_ROOT, + canonicalize_tag, ForumThreadUrl, GardenItemUrl, RelativePath, room_id_from_route_segment, room_route_segment, ROOM_SHORT_ID_LEN, - TildeHttpPathTail, TildeOntologyPath, TildePath, tilde_http_path_to_canonical, + TildeHttpPathTail, TildeOntologyPath, TildePath, tilde_http_path_to_item_id, }; pub use url_normalize::normalize_http_identity_url; diff --git a/types/src/paths.rs b/types/src/paths.rs index 5c60e762c2ca74d79c74237097d5bcc02ba74af2..a1d066d83e7fbcefc27707dd682e54dbcec1cc56 100644 --- a/types/src/paths.rs +++ b/types/src/paths.rs @@ -3,37 +3,23 @@ //! //! ## String kinds (parse in this module only) //! -//! - **[`canonicalize_item`] / [`CanonicalItemUrl`]** — graph storage key and DSL form; tilde +//! - **[`canonicalize_item`] / [`crate::ItemId`]** — graph storage key and DSL form; tilde //! ontology root is always [`SLUG_TILDE_ONTOLOGY_ROOT`] (no `…/~/` trailing slash only). //! - **[`TildeHttpPathTail`]** — capture from `GET /~/*path` or `…/r/{short}{slug}/~/…` (the `*path` segment). //! - **`-/…` wire form** — external items; see [`canonicalize_item`] dash branch. //! - **[`GardenItemUrl`], [`ForumThreadUrl`]** — JSON / browser href surfaces. //! - **[`ROOM_SHORT_ID_LEN`] / [`room_route_segment`]** — `/r/{short}{slug}` vs wire `short/slug`. -use std::borrow::Borrow; use std::fmt; use std::ops::Deref; use serde::{Deserialize, Serialize}; -use crate::url_normalize::{host_preserves_dash_path_case, normalize_http_identity_url}; - -// --------------------------------------------------------------------------- -// Slug tilde ontology (single storage form for `~/`) -// --------------------------------------------------------------------------- - -/// Canonical absolute URL for the tilde ontology **root** (`~/` in UI). Used as the -/// `item_children` parent key for top-level items and must match [`CanonicalItemUrl::ontology_root`]. -pub const SLUG_TILDE_ONTOLOGY_ROOT: &str = "https://slug.social/~"; - -/// Collapse legacy or parser variants of the ontology root to [`SLUG_TILDE_ONTOLOGY_ROOT`]. -pub fn normalize_slug_ontology_storage_url(s: &str) -> String { - if s == "https://slug.social/~/" { - SLUG_TILDE_ONTOLOGY_ROOT.to_string() - } else { - s.to_string() - } -} +use crate::item_id::ItemId; +pub use crate::item_wire::{ + canonicalize_item, item_parent_path, item_path_segments, normalize_slug_ontology_storage_url, + SLUG_TILDE_ONTOLOGY_ROOT, +}; // --------------------------------------------------------------------------- // Private room HTTP path (`/r/{short}{slug}`; wire id remains `short/slug`) @@ -84,310 +70,10 @@ pub fn canonicalize_tag(input: &str) -> String { input.trim().trim_start_matches('#').to_lowercase() } -fn finalize_external_identity_url(s: String) -> String { - if s.starts_with("https://slug.social/") { - return s; - } - let normalized = normalize_http_identity_url(&s).unwrap_or_else(|| s.clone()); - strip_redundant_root_slash(&normalized).unwrap_or(normalized) -} - -/// `url::Url` serializes bare hosts with a `/` path; we keep host-only items slash-free for stable -/// keys matching the pre-normalizer spellings. -fn strip_redundant_root_slash(s: &str) -> Option { - let u = url::Url::parse(s).ok()?; - if u.path() == "/" && u.query().is_none() && u.fragment().is_none() { - let scheme = u.scheme(); - let host = u.host_str()?; - return Some(match u.port() { - Some(p) => format!("{scheme}://{host}:{p}"), - None => format!("{scheme}://{host}"), - }); - } - None -} - -/// Ontology item reference → canonical absolute URL on the slug host. -pub fn canonicalize_item(input: &str) -> String { - let s = input.trim(); - if s.is_empty() { - return String::new(); - } - - // External scope: `-/host/path` is the universal alias for `https://host/path`. - if let Some(rest) = s.strip_prefix("-/") { - let (host, tail) = rest - .split_once('/') - .map_or((rest, ""), |(h, t)| (h, t)); - let host = host.trim().to_lowercase(); - if host.is_empty() { - return String::new(); - } - let preserve_case = host_preserves_dash_path_case(&host); - return if tail.is_empty() { - finalize_external_identity_url(format!("https://{}", host)) - } else { - let path = tail - .trim_start_matches('/') - .trim_end_matches('/') - .split('/') - .filter_map(|seg| { - let t = seg.trim(); - if t.is_empty() { - None - } else if preserve_case { - Some(t.to_string()) - } else { - Some(t.to_lowercase()) - } - }) - .collect::>() - .join("/"); - finalize_external_identity_url(format!("https://{}/{}", host, path)) - }; - } - - if let Some(rest) = s.strip_prefix("https://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let host = host.trim().to_lowercase(); - return finalize_external_identity_url(if tail.is_empty() { - format!("https://{}", host) - } else { - format!("https://{}/{}", host, tail) - }); - } - if let Some(rest) = s.strip_prefix("http://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let host = host.trim().to_lowercase(); - return finalize_external_identity_url(if tail.is_empty() { - format!("http://{}", host) - } else { - format!("http://{}/{}", host, tail) - }); - } - - let is_tilde = s.starts_with("~/"); - let rest = s.strip_prefix("~/").or_else(|| s.strip_prefix("/")).unwrap_or(s); - - let tail = rest - .split('/') - .filter_map(|seg| { - let t = seg.trim(); - if t.is_empty() { - None - } else { - Some(t.to_lowercase()) - } - }) - .collect::>() - .join("/"); - - if is_tilde { - if tail.is_empty() { - return SLUG_TILDE_ONTOLOGY_ROOT.to_string(); - } - format!("https://slug.social/~/{}", tail) - } else if tail.is_empty() { - "https://slug.social".to_string() - } else { - format!("https://slug.social/{}", tail) - } -} - -pub fn item_path_segments(input: &str) -> Vec { - let canonical = canonicalize_item(input); - if canonical.is_empty() { - return vec![]; - } - - if let Some(rest) = canonical.strip_prefix("https://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let mut out = vec![format!("https://{}", host)]; - out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); - return out; - } - if let Some(rest) = canonical.strip_prefix("http://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let mut out = vec![format!("http://{}", host)]; - out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); - return out; - } - - canonical - .split('/') - .filter(|s| !s.is_empty()) - .map(|s| s.to_string()) - .collect() -} - -pub fn item_parent_path(input: &str) -> Option { - let segs = item_path_segments(input); - if segs.len() <= 1 { - return None; - } - Some(segs[..segs.len() - 1].join("/")) -} - -fn external_display_dash_prefix(host_and_path: &str) -> String { - let (host, path) = host_and_path - .split_once('/') - .map_or((host_and_path, ""), |(h, p)| (h, p)); - let host = host.trim().to_lowercase(); - let path = path - .trim_end_matches('/') - .split('/') - .filter_map(|seg| { - let t = seg.trim(); - if t.is_empty() { - None - } else { - Some(t.to_lowercase()) - } - }) - .collect::>() - .join("/"); - if path.is_empty() { - format!("-/{}", host) - } else { - format!("-/{}", format!("{}/{}", host, path)) - } -} - // --------------------------------------------------------------------------- // Storage + input path newtypes // --------------------------------------------------------------------------- -/// Canonical item identifier as produced by [`canonicalize_item`]. -/// -/// Shared across all scopes; room is not embedded. Usually -/// `https://slug.social/~/…` or an external `http(s)://…` URL item. -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -pub struct CanonicalItemUrl(pub String); - -impl CanonicalItemUrl { - pub fn parse(input: &str) -> Option { - let c = canonicalize_item(input); - if c.is_empty() { - None - } else { - Some(Self(normalize_slug_ontology_storage_url(&c))) - } - } - - pub fn as_str(&self) -> &str { - &self.0 - } - - /// Collapses legacy slug ontology root spellings so [`HashMap`] keys match the reducer graph. - pub fn normalized_storage(self) -> Self { - Self(normalize_slug_ontology_storage_url(self.as_str())) - } - - pub fn tilde_tail(&self) -> Option<&str> { - if let Some(tail) = self.0.strip_prefix("https://slug.social/~/") { - return Some(tail); - } - if self.0 == SLUG_TILDE_ONTOLOGY_ROOT || self.0 == "https://slug.social/~/" { - return Some(""); - } - None - } - - pub fn last_segment(&self) -> &str { - self.0 - .rsplit('/') - .find(|s| !s.is_empty()) - .unwrap_or(self.0.as_str()) - } - - pub fn ontology_root() -> Self { - Self(SLUG_TILDE_ONTOLOGY_ROOT.to_string()) - } - - pub fn parent(&self) -> Option { - if self.tilde_tail().is_some() { - if self.tilde_tail().map(|t| t.is_empty()).unwrap_or(true) { - return None; - } - let last_slash = self.0.rfind('/')?; - let parent_str = &self.0[..last_slash]; - if parent_str.is_empty() { - None - } else { - Some(Self(parent_str.to_string())) - } - } else if let Some(rest) = self.0.strip_prefix("https://") { - Self::parent_http_url("https://", rest) - } else if let Some(rest) = self.0.strip_prefix("http://") { - Self::parent_http_url("http://", rest) - } else { - None - } - } - - fn parent_http_url(scheme: &'static str, rest: &str) -> Option { - let (host, path) = rest.split_once('/').map_or((rest, ""), |(h, p)| (h, p)); - let host = host.trim(); - let path = path.trim_end_matches('/'); - if path.is_empty() { - return None; - } - let parent_path = path.rsplit_once('/').map(|(p, _)| p).unwrap_or(""); - if parent_path.is_empty() { - Some(Self(format!("{scheme}{}", host))) - } else { - Some(Self(format!("{scheme}{}/{}", host, parent_path))) - } - } - - /// `-/` representation for external `https://…` items, `~/…` for slug ontology, else unchanged. - pub fn display_path(&self) -> String { - if let Some(tail) = self.tilde_tail() { - if tail.is_empty() { - return "~/".to_string(); - } - return format!("~/{}", tail); - } - if let Some(tail) = self.0.strip_prefix("https://") { - if tail.starts_with("slug.social") { - self.0.clone() - } else { - external_display_dash_prefix(tail) - } - } else if let Some(tail) = self.0.strip_prefix("http://") { - if tail.starts_with("slug.social") { - self.0.clone() - } else { - external_display_dash_prefix(tail) - } - } else { - self.0.clone() - } - } - - pub fn tilde_segments(&self) -> Vec<&str> { - match self.tilde_tail() { - Some(tail) if !tail.is_empty() => { - std::iter::once("~") - .chain(tail.split('/').filter(|s| !s.is_empty())) - .collect() - } - Some(_) => vec!["~"], - None => vec![], - } - } - - /// `~/…` list label for ontology items (paths index, CLI). - pub fn tilde_list_label(&self) -> TildeOntologyPath { - TildeOntologyPath::from_stored(self) - } - - /// Absolute href for JSON/RPC and browsers for this stored id in `room`. - pub fn json_href(&self, room_wire: &str) -> GardenItemUrl { - GardenItemUrl::from_stored(self, room_wire) - } -} - /// HTTP route capture: path segment after `~/` in `GET /~/*path` or `…/r/{short}{slug}/~/…` (empty = ontology root). #[derive(Debug, Clone, PartialEq, Eq, Hash)] pub struct TildeHttpPathTail(pub String); @@ -401,13 +87,13 @@ impl TildeHttpPathTail { &self.0 } - pub fn to_canonical(&self) -> CanonicalItemUrl { - tilde_http_path_to_canonical(self.as_str()) + pub fn to_item_id(&self) -> ItemId { + tilde_http_path_to_item_id(self.as_str()) } } -/// Map the router's tilde tail (e.g. `topic/a`, or empty for root) to a [`CanonicalItemUrl`]. -pub fn tilde_http_path_to_canonical(path_segment: &str) -> CanonicalItemUrl { +/// Map the router's tilde tail (e.g. `topic/a`, or empty for root) to an [`ItemId`]. +pub fn tilde_http_path_to_item_id(path_segment: &str) -> ItemId { let p = path_segment.trim_start_matches('/'); let raw = if p.starts_with("http://") || p.starts_with("https://") { p.to_string() @@ -416,37 +102,7 @@ pub fn tilde_http_path_to_canonical(path_segment: &str) -> CanonicalItemUrl { } else { format!("~/{}", p) }; - CanonicalItemUrl::parse(&raw).unwrap_or_else(|| CanonicalItemUrl::ontology_root()) -} - -impl fmt::Display for CanonicalItemUrl { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - self.0.fmt(f) - } -} - -impl Borrow for CanonicalItemUrl { - fn borrow(&self) -> &str { - &self.0 - } -} - -impl PartialEq for CanonicalItemUrl { - fn eq(&self, other: &str) -> bool { - self.0 == other - } -} - -impl PartialEq<&str> for CanonicalItemUrl { - fn eq(&self, other: &&str) -> bool { - self.0 == *other - } -} - -impl PartialEq for CanonicalItemUrl { - fn eq(&self, other: &String) -> bool { - &self.0 == other - } + ItemId::parse(&raw).unwrap_or_else(ItemId::ontology_root) } #[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] @@ -468,8 +124,8 @@ impl TildePath { &self.0 } - pub fn canonicalize(&self) -> Option { - CanonicalItemUrl::parse(&self.0) + pub fn canonicalize(&self) -> Option { + ItemId::parse(&self.0) } } @@ -496,7 +152,7 @@ impl RelativePath { &self.0 } - pub fn join_under_ontology_root(&self, root: &CanonicalItemUrl) -> Option { + pub fn join_under_ontology_root(&self, root: &ItemId) -> Option { let base = root.tilde_tail()?; let joined = if base.is_empty() { if self.0.is_empty() { @@ -509,7 +165,7 @@ impl RelativePath { } else { format!("~/{}/{}", base.trim_end_matches('/'), self.0) }; - CanonicalItemUrl::parse(&joined) + ItemId::parse(&joined) } } @@ -545,14 +201,17 @@ impl GardenItemUrl { self.0 } - /// Stored canonical id + RPC `room` field (`"public"` or `"short/slug"`). - pub fn from_stored(stored: &CanonicalItemUrl, room_wire: &str) -> Self { - Self(garden_href_string(stored.as_str(), room_wire)) + /// Stored item id + RPC `room` field (`"public"` or `"short/slug"`). + pub fn from_stored(stored: &ItemId, room_wire: &str) -> Self { + Self(garden_href_string(stored, room_wire)) } /// Like [`Self::from_stored`] but accepts a string that may already be canonical. pub fn from_storage_str(stored: &str, room_wire: &str) -> Self { - Self(garden_href_string(stored, room_wire)) + let Some(id) = ItemId::parse(stored) else { + return Self(api_path_or_url(stored)); + }; + Self(garden_href_string(&id, room_wire)) } } @@ -570,18 +229,15 @@ impl Deref for GardenItemUrl { } } -fn garden_href_string(item: &str, room_wire: &str) -> String { +fn garden_href_string(c: &ItemId, room_wire: &str) -> String { let room = room_wire.trim(); if room.is_empty() || room == "public" { - return api_path_or_url(item); + return api_path_or_url(c.as_str()); } let Some(room_seg) = room_route_segment(room) else { - return api_path_or_url(item); - }; - let Some(c) = CanonicalItemUrl::parse(item) else { - return api_path_or_url(item); + return api_path_or_url(c.as_str()); }; - let root = CanonicalItemUrl::ontology_root(); + let root = ItemId::ontology_root(); let item_norm = c.as_str().trim_end_matches('/'); let root_norm = root.as_str().trim_end_matches('/'); if let Some(tail) = c.tilde_tail() { @@ -600,7 +256,7 @@ fn garden_href_string(item: &str, room_wire: &str) -> String { let tail = tail.strip_prefix("-/").unwrap_or(tail.as_str()); return format!("https://slug.social/r/{room_seg}/-/{tail}"); } - api_path_or_url(item) + api_path_or_url(c.as_str()) } /// Forum thread URL for JSON (`/t/…` or `/r/…/t/…` on slug.social). @@ -650,7 +306,7 @@ impl Deref for ForumThreadUrl { pub struct TildeOntologyPath(pub String); impl TildeOntologyPath { - pub fn from_stored(c: &CanonicalItemUrl) -> Self { + pub fn from_stored(c: &ItemId) -> Self { Self(c.display_path()) } @@ -679,22 +335,22 @@ mod tests { #[test] fn canonical_parent_deep() { - let c = CanonicalItemUrl::parse("~/a/b/c").unwrap(); + let c = ItemId::parse("~/a/b/c").unwrap(); assert_eq!(c.parent().unwrap().as_str(), "https://slug.social/~/a/b"); } #[test] fn canonical_parent_one_level() { - let c = CanonicalItemUrl::parse("~/a").unwrap(); + let c = ItemId::parse("~/a").unwrap(); assert_eq!(c.parent().unwrap().as_str(), "https://slug.social/~"); } #[test] fn canonical_parent_root_is_none() { - let root = CanonicalItemUrl::parse("~/").unwrap(); + let root = ItemId::parse("~/").unwrap(); assert!(root.parent().is_none()); assert_eq!(root.as_str(), SLUG_TILDE_ONTOLOGY_ROOT); - assert_eq!(root, CanonicalItemUrl::ontology_root()); + assert_eq!(root, ItemId::ontology_root()); } #[test] @@ -705,49 +361,49 @@ mod tests { SLUG_TILDE_ONTOLOGY_ROOT.to_string() ); assert_eq!( - CanonicalItemUrl::parse("https://slug.social/~/") + ItemId::parse("https://slug.social/~/") .unwrap() .as_str(), SLUG_TILDE_ONTOLOGY_ROOT ); - let legacy = CanonicalItemUrl("https://slug.social/~/".to_string()); + let legacy = ItemId::opaque("https://slug.social/~/".to_string()); assert_eq!(legacy.normalized_storage().as_str(), SLUG_TILDE_ONTOLOGY_ROOT); } #[test] fn tilde_http_path_tail_maps_router_segment() { assert_eq!( - TildeHttpPathTail::new("").to_canonical(), - CanonicalItemUrl::ontology_root() + TildeHttpPathTail::new("").to_item_id(), + ItemId::ontology_root() ); assert_eq!( - tilde_http_path_to_canonical("topic/x").as_str(), + tilde_http_path_to_item_id("topic/x").as_str(), "https://slug.social/~/topic/x" ); } #[test] fn display_path_slug_ontology_root() { - let r = CanonicalItemUrl::ontology_root(); + let r = ItemId::ontology_root(); assert_eq!(r.display_path(), "~/"); assert_eq!(r.tilde_tail(), Some("")); } #[test] fn tilde_segments_deep() { - let c = CanonicalItemUrl::parse("~/a/b").unwrap(); + let c = ItemId::parse("~/a/b").unwrap(); assert_eq!(c.tilde_segments(), vec!["~", "a", "b"]); } #[test] fn tilde_segments_root() { - let c = CanonicalItemUrl::parse("~/").unwrap(); + let c = ItemId::parse("~/").unwrap(); assert_eq!(c.tilde_segments(), vec!["~"]); } #[test] fn tilde_segments_non_ontology_is_empty() { - let c = CanonicalItemUrl::parse("https://example.com/foo").unwrap(); + let c = ItemId::parse("https://example.com/foo").unwrap(); assert_eq!(c.tilde_segments(), Vec::<&str>::new()); } @@ -831,28 +487,28 @@ mod tests { } #[test] - fn canonical_item_url_parent_external_strips_last_segment() { - let c = CanonicalItemUrl::parse("https://spotify.com/track/1").unwrap(); + fn item_id_parent_external_strips_last_segment() { + let c = ItemId::parse("https://spotify.com/track/1").unwrap(); assert_eq!( c.parent().unwrap().as_str(), "https://spotify.com/track" ); assert_eq!( - CanonicalItemUrl::parse("https://github.com/iss/1") + ItemId::parse("https://github.com/iss/1") .unwrap() .parent() .unwrap() .as_str(), "https://github.com/iss" ); - assert!(CanonicalItemUrl::parse("https://github.com").unwrap().parent().is_none()); + assert!(ItemId::parse("https://github.com").unwrap().parent().is_none()); } #[test] fn display_path_roundtrips_dash_and_tilde() { - let ext = CanonicalItemUrl::parse("https://GitHub.com/org/Issue").unwrap(); + let ext = ItemId::parse("https://GitHub.com/org/Issue").unwrap(); assert_eq!(ext.display_path(), "-/github.com/org/issue"); - let tilde = CanonicalItemUrl::parse("~/Rust/Doc").unwrap(); + let tilde = ItemId::parse("~/Rust/Doc").unwrap(); assert_eq!(tilde.display_path(), "~/rust/doc"); }