constitution · epochs · watch · epoch 3

comparison

c_f10e7b043e68 (tommy-mor) vs c_6f04dcb2e38c (tommy-mor)

download prompt · raw event · cmp_2df1451ab9fb08

council reasoning

~anthropic/claude-sonnet-latest · winner A · 6:4 · permalink

Side A replaces an ad-hoc, per-domain if/else URL normalizer with a cleaner, more principled graph-based parser (DFA + generic fallback), improving the actual product's core canonicalization logic while preserving test coverage intent. Side B is mostly internal admin/debug page cleanup (removing legacy projection code paths and adjusting tests/docs), which is useful maintenance but lower-impact and more narrowly scoped to internal tooling.

~x-ai/grok-latest · winner A · 2:1 · permalink

Side A replaces ad-hoc URL combinators (engine.rs ParsedUrl pipeline and per-domain normalize_* functions) with a graph/DFA-based canonicalization API wired through registry (graph(), resolve_canonical, breadcrumbs), which is lasting core domain design. Side B is valuable cleanup—dropping _legacy_commit_row/_legacy_observation and GitDiscovery projections so epoch/commit pages use Evidence only—but it mainly removes compatibility paths and tightens Emission fields rather than adding new capability.

openai/gpt-chat-latest · winner B · 4:1 · permalink

Side B makes a substantive architectural cleanup by removing legacy GitDiscovery projection paths, requiring evidence metadata in `Emission`, simplifying epoch/commit endpoints to use Evidence envelopes only, and updating tests to enforce the new model. Side A is primarily a refactor that replaces the old URL engine with new graph/parse modules and updates documentation and tests, but the patch shown mostly removes the previous implementation while delegating behavior elsewhere, making its lasting functional impact less directly evident.

sides

A — c_f10e7b043e68 (tommy-mor)

message

[7bb7145d] url stuff

diff preview

diff --git a/AGENTS.md b/AGENTS.md
index e60b9ba6012593361ef10e8fdd9439cd9932e09b..babb889d6fbfb1fa7176c9e6b7544ae17b61dd2e 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -58,4 +58,4 @@ Use **tmux** for `cargo run --package sorter2-server` (dev server). Rebuild afte
 
 - First `cargo test` / `cargo build --release` is slow; Clojure smoke test always does a release build.
 - `legacy/` and `ideas/` are not part of the workspace build.
-- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/`](server/src/url_rules/) (composable Rust, not a config DSL). After changing canonicalization rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
+- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/graph.rs`](server/src/url_rules/graph.rs): a semantic graph (DFA on host + path, query params in `Context`) with a generic internet fallback for unknown sites. After changing rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
diff --git a/server/src/url_rules/engine.rs b/server/src/url_rules/engine.rs
deleted file mode 100644
index e29b6b48c08deb7bffe031b1e542b1e25a7bef15..0000000000000000000000000000000000000000
--- a/server/src/url_rules/engine.rs
+++ /dev/null
@@ -1,187 +0,0 @@
-//! Composable URL normalization primitives.
-
-use std::collections::HashMap;
-
-use url::Url;
-
-/// Mutable URL view used by rule combinators before serializing to a canonical string.
-#[derive(Debug, Clone)]
-pub struct ParsedUrl {
-    pub scheme: String,
-    pub host: String,
-    pub path_segments: Vec<String>,
-    pub query: HashMap<String, String>,
-    pub fragment: Option<String>,
-}
-
-impl ParsedUrl {
-    pub fn parse(raw: &str) -> Option<Self> {
-        let trimmed = raw.trim();
-        if trimmed.is_empty() {
-            return None;
-        }
-
-        let with_scheme = if trimmed.contains("://") {
-            trimmed.to_string()
-        } else if trimmed.starts_with("r/") || trimmed.starts_with("/r/") {
-            let rest = trimmed.trim_start_matches('/').trim_start_matches("r/");
-            format!("https://reddit.com/r/{rest}")
-        } else if trimmed.contains('.') && !trimmed.starts_with('/') {
-            format!("https://{trimmed}")
-        } else {
-            trimmed.to_string()
-        };
-
-        let url = Url::parse(&with_scheme).ok()?;
-        let host = url.host_str()?.to_string();
-        let path_segments: Vec<String> = url
-            .path_segments()
-            .map(|segs| segs.filter(|s| !s.is_empty()).map(str::to_string).collect())
-            .unwrap_or_default();
-
-        let mut query = HashMap::new();
-        for (k, v) in url.query_pairs() {
-            query.insert(k.into_owned(), v.into_owned());
-        }
-
-        Some(Self {
-            scheme: url.scheme().to_string(),
-            path_segments,
-            query,
-            fragment: url.fragment().map(str::to_string),
-            host,
-        })
-    }
-
-    pub fn with_path_segments(&self, segments: &[String]) -> Self {
-        let mut u = self.clone();
-        u.path_segments = segments.to_vec();
-        u
-    }
-
-    pub fn to_url(&self) -> Option<Url> {
-        let mut url = if self.path_segments.is_empty() {
-            Url::parse(&format!("{}://{}", self.scheme, self.host)).ok()?
-        } else {
-            let path = format!("/{}", self.path_segments.join("/"));
-            Url::parse(&format!("{}://{}{}", self.scheme, self.host, path)).ok()?
-        };
-        if !self.query.is_empty() {
-            let mut pairs: Vec<_> = self.query.iter().collect();
-            pairs.sort_by(|a, b| a.0.cmp(b.0));
-            url.query_pairs_mut().clear();
-            for (k, v) in pairs {
-                url.query_pairs_mut().append_pair(k, v);
-            }
-        }
-        if let Some(ref frag) = self.fragment {
-            url.set_fragment(Some(frag));
-        }
-        Some(url)
-    }
-
-    pub fn canonical_string(&self) -> Option<String> {
-        let url = self.to_url()?;
-        let mut s = url.to_string();
-        if self.path_segments.is_empty() {
-            s = s.trim_end_matches('/').to_string();
-        }
-        Some(s)
-    }
-}
-
-pub fn force_https(u: &mut ParsedUrl) {
-    if u.scheme == "http" {
-        u.scheme = "https".to_string();
-    }
-}
-
-pub fn drop_fragment(u: &mut ParsedUrl) {
-    u.fragment = None;
-}
-
-pub fn strip_www(u: &mut ParsedUrl) {
-    if u.host.starts_with("www.") {
-        u.host = u.host[4..].to_string();
-    }
-}
-
-pub fn lowercase_host(u: &mut ParsedUrl) {
-    u.host = u.host.to_ascii_lowercase();
-}
-
-pub fn lowercase_path(u: &mut ParsedUrl) {
-    for seg in &mut u.path_segments {
-        *seg = seg.to_ascii_lowercase();
-    }
-}
-
-pub fn clear_query(u: &mut ParsedUrl) {
-    u.query.clear();
-}
-
-pub fn keep_only_query(u: &mut ParsedUrl, keys: &[&str]) {
-    u.query
-        .retain(|k, _| keys.iter().any(|want| want == &k.as_str()));
-}
-
-pub fn strip_tracking_params(u: &mut ParsedUrl) {
-    u.query.retain(|k, _| {
-        let lower = k.to_ascii_lowercase();
-        !(lower.starts_with("utm_")
-            || matches!(
-                lower.as_str(),
-                "fbclid" | "gclid" | "ref" | "ref_src" | "ref_source" | "mc_cid" | "mc_eid"
-            ))
-    });
-}
-
-pub fn truncate_after_segment(u: &mut ParsedUrl, name: &str, keep: usize) {
-    if let Some(i) = u.path_segments.iter().position(|s| s == name) {
-        let end = (i + 1 + keep).min(u.path_segments.len());
-        u.path_segments.truncate(end);
-    }
-}
-
-pub fn drop_listing_suffix(u: &mut ParsedUrl, suffixes: &[&str]) {
-    if u.path_segments.len() >= 3 && u.path_segments.first().map(String::as_str) == Some("r") {
-        if let Some(last) = u.path_segments.last() {
-            if suffixes.iter().any(|s| *s == last.as_str()) {
-                u.path_segments.pop();
-            }
-        }
-    }
-}
-
-pub fn normalize_reddit_host(u: &mut ParsedUrl) {
-    if matches!(
-        u.host.as_str(),
-        "old.reddit.com" | "new.reddit.com" | "www.reddit.com"
-    ) {
-        u.host = "reddit.com".to_string();
-    }
-}
-
-pub fn rewrite_youtu_be(u: &mut ParsedUrl) {
-    if u.host == "youtu.be" && u.path_segments.len() == 1 {
-        let id = u.path_segments[0].clone();
-        u.host = "youtube.com".to_string();
-        u.path_segments = vec!["watch".to_string()];
-        u.query.insert("v".to_string(), id);
-    }
-}
-
-pub fn rewrite_youtube_shorts(u: &mut ParsedUrl) {
-    if u.host == "youtube.com" && u.path_segments.first().map(String::as_str) == Some("shorts") {
-        if let Some(id) = u.path_segments.get(1).cloned() {
-            u.path_segments = vec!["watch".to_string()];
-            u.query.insert("v".to_string(), id);
-        }
-    }
-}
-
-pub fn normalize_youtube_host(u: &mut ParsedUrl) {
-    if matches!(u.host.as_str(), "m.youtube.com" | "www.youtube.com") {
-        u.host = "youtube.com".to_string();
-    }
-}
diff --git a/server/src/url_rules/mod.rs b/server/src/url_rules/mod.rs
index 03d53bd3e82d704a01ba3fd8dd02b7d31422c0de..9e1445346ce77a49dd6a7e7713bf9c57aef353cc 100644
--- a/server/src/url_rules/mod.rs
+++ b/server/src/url_rules/mod.rs
@@ -1,8 +1,12 @@
-//! URL canonicalization and hierarchy rules for [`crate::path_types::ItemId`].
+//! URL canonicalization and hierarchy via a semantic graph (DFA + generic fallback).
 
-mod engine;
+mod graph;
+mod parse;
 mod registry;
 
+#[cfg(test)]
+mod registry_tests;
+
 pub use registry::{
     canonicalize_raw, looks_like_url, navigable_breadcrumbs, parent_url, resolve_id, CanonicalResult,
 };
diff --git a/server/src/url_rules/registry.rs b/server/src/url_rules/registry.rs
index 14514e9af8385fb2b9b2f35eb9ee14d453d4b97c..8e6c012ea1fc74b864307bdacdf5a0f5db5259fc 100644
--- a/server/src/url_rules/registry.rs
+++ b/server/src/url_rules/registry.rs
@@ -1,12 +1,7 @@
-//! Per-domain canonicalization and hierarchy rules.
+//! Public API: canonical identity and hierarchy via the URL graph.
 
-use std::collections::HashSet;
-
-use super::engine::{
-    clear_query, drop_fragment, drop_listing_suffix, force_https, keep_only_query, lowercase_host,
-    lowercase_path, normalize_reddit_host, normalize_youtube_host, rewrite_youtu_be,
-    rewrite_youtube_shorts, strip_tracking_params, strip_www, truncate_after_segment, ParsedUrl,
-};
+use super::graph::graph;
+use super::parse::UrlParts;
 
 /// Result of canonicalizing a raw URL string.
 #[derive(Debug, Clone, PartialEq, Eq)]
@@ -16,71 +11,16 @@ pub struct CanonicalResult {
     pub alias_of: Option<String>,
 }
 
-fn apply_global(u: &mut ParsedUrl) {
-    force_https(u);
-    drop_fragment(u);
-    strip_www(u);
-    lowercase_host(u);
-    strip_tracking_params(u);
-}
-
-fn normalize_reddit(u: &mut ParsedUrl) {
-    normalize_reddit_host(u);
-    lowercase_path(u);
-    truncate_after_segment(u, "comments", 1);
-    drop_listing_suffix(u, &["hot", "top", "new", "rising", "controversial"]);
-    clear_query(u);
-}
-
-fn normalize_youtube(u: &mut ParsedUrl) {
-    rewrite_youtu_be(u);
-    normalize_youtube_host(u);
-    rewrite_youtube_shorts(u);
-    keep_only_query(u, &["v", "list"]);
-}
-
-fn normalize_default(_u: &mut ParsedUrl) {
-    // Global rules only.
-}
-
-fn domain_key(host: &str) -> &'static str {
-    if host == "reddit.com" || host.ends_with(".reddit.com") {
-        "reddit.com"
-    } else if host == "youtube.com" || host == "youtu.be" {
-        "youtube.com"
-    } else {
-        "default"
-    }
-}
-
-fn normalize_for_host(u: &mut ParsedUrl) {
-    apply_global(u);
-    match domain_key(&u.host) {
-        "reddit.com" => normalize_reddit(u),
-        "youtube.com" => normalize_youtube(u),
-        _ => normalize_default(u),
-    }
-}
-
-/// Structural path segments that must not become standalone tree nodes when more path follows.
-fn structural_trailing(host: &str) -> &'static [&'static str] {
-    match domain_key(host) {
-        "reddit.com" => &["comments"],
-        _ => &[],
-    }
-}
-
 /// Canonicalize a raw URL. Returns `None` if the input is not URL-like.
 pub fn canonicalize_raw(raw: &str) -> Option<CanonicalResult> {
     let trimmed = raw.trim();
     if trimmed.is_empty() {
         return None;
     }
-    let mut u = ParsedUrl::parse(trimmed)?;
-    let input_snapshot = u.canonical_string()?;
-    normalize_for_host(&mut u);
-    let canonical = u.canonical_string()?;
-    let alias_of = if input_snapshot != canonical {
+    let parts = UrlParts::parse(trimmed)?;
+    let g = graph();
+    let canonical = g.resolve_canonical(&parts)?;
+    let alias_of = if trimmed != canonical {
         Some(trimmed.to_string())
     } else {
         None
@@ -98,35 +38,14 @@ pub fn resolve_id(raw: &str) -> Option<String> {
 
 /// Navigable ancestor URLs from domain root up to and including `canonical` (full URLs).
 pub fn navigable_breadcrumbs(canonical: &str) -> Vec<String> {
-    let Some(u) = ParsedUrl::parse(canonical) else {
-        return vec![canonical.to_string()];
+    let parts = match UrlParts::parse(canonical) {
+        Some(p) => p,
+        None => return vec![canonical.to_string()],
     };
-    let structural: HashSet<&str> = structural_trailing(&u.host).iter().copied().collect();
-    let n = u.path_segments.len();
-    let mut out = Vec::new();
-
-    // Domain root (no path segments).
-    if let Some(base) = u.with_path_segments(&[]).canonical_string() {
-        out.push(base);
-    }
-
-    for i in 0..n {
-        let segs: Vec<String> = u.path_segments[..=i].to_vec();
-        let is_last = i == n - 1;
-        let seg = u.path_segments[i].as_str();
-        if structural.contains(seg) && !is_last {
-            continue;
-        }
-        if let Some(url) = u.with_path_segments(&segs).canonical_string() {
-            if out.last() != Some(&url) {
- 

… preview truncated; 3,144 characters omitted

download full diff A

B — c_6f04dcb2e38c (tommy-mor)

message

[bda5f8aa] Remove legacy evidence projection; Evidence envelopes only.

Epoch and commit pages no longer invent history from bare GitDiscovery rows. Production ledger will be wiped to re-emit under the current schema.

Co-authored-by: Cursor <cursoragent@cursor.com>

diff preview

diff --git a/constitution.py b/constitution.py
index 4dd5b9dfbba231d46289c490f91b1dd5b1018bcf..26ba130e885e0e69fb7874ca5c3f07f42100a150 100644
--- a/constitution.py
+++ b/constitution.py
@@ -210,10 +210,10 @@ class Emission:
     distributions: dict   # author -> amount str
     ranking: dict         # author -> score str
     models_used: list
-    discovery_snapshot_id: str = ""  # empty only for pre-discovery ledger history
-    evidence_schema_version: int = 1
-    ranking_run_id: str = ""
-    ranking_event_id: str = ""
+    discovery_snapshot_id: str
+    evidence_schema_version: int
+    ranking_run_id: str
+    ranking_event_id: str
 
 
 @event
@@ -502,26 +502,6 @@ def _epochs_in_ledger() -> list[int]:
     return sorted(epochs)
 
 
-def _legacy_commit_row(commit_id: str) -> tuple[GitDiscovery | None, dict | None]:
-    for discovery in store.read():
-        if not isinstance(discovery, GitDiscovery):
-            continue
-        for commit in discovery.commits:
-            if commit_id_for_oid(commit["oid"]) == commit_id:
-                return discovery, commit
-    return None, None
-
-
-def _legacy_observation(commit_id: str) -> tuple[GitDiscovery | None, dict | None]:
-    for discovery in store.read():
-        if not isinstance(discovery, GitDiscovery):
-            continue
-        for obs in discovery.observations:
-            if commit_id_for_oid(obs["oid"]) == commit_id:
-                return discovery, obs
-    return None, None
-
-
 def build_pairwise_prompt(side_a: dict, side_b: dict) -> str:
     return f"""You are ranking contributions to an open source project.
 Compare these two sides (each may be one or more commits). Decide which side contributed more.
@@ -2040,7 +2020,7 @@ def _strip_heavy_fields(obj: dict) -> dict:
 
 @app.get("/api/ledger")
 async def get_ledger(offset: int = 0, limit: int = 100, full: int = 0):
-    """List of ledger dicts (backward-compatible). Heavy blobs stripped unless full=1."""
+    """List of ledger dicts. Heavy blobs stripped unless full=1."""
     limit = max(1, min(limit, 500))
     rows = []
     for e in store.read()[offset:offset + limit]:
@@ -2274,23 +2254,25 @@ async def epochs_index():
     epochs = _epochs_in_ledger()
     rows = []
     for epoch in epochs:
-        discovery = _discovery_for_epoch(epoch)
         emission = _emission_for_epoch(epoch)
-        evidence_n = sum(
-            1 for e in evidence_by_kind() if e.epoch == epoch
+        evidence_n = sum(1 for e in evidence_by_kind() if e.epoch == epoch)
+        disc = next(
+            (
+                e for e in evidence_by_kind("git.discovery_completed")
+                if e.epoch == epoch
+            ),
+            None,
         )
         detail = []
-        if discovery:
+        if disc:
             detail.append(
-                f"{len(discovery.commits)} eligible / "
-                f"{len(discovery.observations)} observed"
+                f"{disc.payload.get('eligible_count', 0)} eligible / "
+                f"{disc.payload.get('observation_count', 0)} observed"
             )
         if emission:
             detail.append(f"emitted {emission.total_emitted}")
         if evidence_n:
             detail.append(f"{evidence_n} evidence events")
-        elif discovery or emission:
-            detail.append("legacy (no Evidence envelopes)")
         rows.append(["li",
             _a(_evidence_path("epoch", str(epoch)), f"epoch {epoch}"),
             " — ",
@@ -2307,37 +2289,37 @@ async def epochs_index():
 
 @app.get("/epochs/{epoch}")
 async def epoch_detail(epoch: int):
-    discovery = _discovery_for_epoch(epoch)
     emission = _emission_for_epoch(epoch)
     evidence_rows = [e for e in evidence_by_kind() if e.epoch == epoch]
+    if not evidence_rows and emission is None:
+        return _evidence_page(f"epoch {epoch}", [
+            _evidence_nav(),
+            ["h1", f"epoch {epoch}"],
+            ["p.note", "No evidence for this epoch."],
+        ])
+
     commit_evs = [e for e in evidence_rows if e.kind == "git.commit"]
     comparison_evs = [e for e in evidence_rows if e.kind == "comparison.input"]
     judgment_evs = [e for e in evidence_rows if e.kind == "llm.judgment"]
+    discovery_ev = next(
+        (e for e in evidence_rows if e.kind == "git.discovery_completed"), None
+    )
     ranking_started = next(
         (e for e in evidence_rows if e.kind == "ranking.started"), None
     )
     ranking_completed = next(
         (e for e in evidence_rows if e.kind == "ranking.completed"), None
     )
-    legacy = not evidence_rows and (discovery is not None or emission is not None)
-
-    commit_links: list[tuple[str, str]] = []
-    if commit_evs:
-        for e in commit_evs:
-            cid = e.payload.get("commit_id") or ""
-            label = (
-                f"{e.payload.get('oid', cid)[:24]} "
-                f"({e.payload.get('contributor', '?')})"
-            )
-            commit_links.append((label, _evidence_path("commit", cid)))
-    elif discovery:
-        for c in discovery.commits:
-            cid = commit_id_for_oid(c["oid"])
-            commit_links.append((
-                f"{c['oid'][:24]} ({c.get('contributor', '?')})",
-                _evidence_path("commit", cid),
-            ))
 
+    commit_links = [
+        (
+            f"{e.payload.get('oid', e.payload.get('commit_id', ''))[:24]} "
+            f"({e.payload.get('contributor', '?')})",
+            _evidence_path("commit", e.payload["commit_id"]),
+        )
+        for e in commit_evs
+        if e.payload.get("commit_id")
+    ]
     comparison_links = [
         (
             e.payload.get("summary") or e.payload.get("comparison_id", e.event_id),
@@ -2360,16 +2342,13 @@ async def epoch_detail(epoch: int):
     ]
 
     excluded = []
-    if discovery:
-        for obs in discovery.observations:
+    if discovery_ev:
+        for obs in discovery_ev.payload.get("observations") or []:
             if obs.get("eligible"):
                 continue
             oid = obs.get("oid", "?")
             reason = obs.get("exclusion_reason") or "excluded"
-            excluded.append(["li",
-                f"{oid[:28]} — {reason} — ",
-                ["span.note", "legacy evidence unavailable"],
-            ])
+            excluded.append(["li", f"{oid[:28]} — {reason}"])
 
     ranking_nodes: list = []
     if ranking_completed:
@@ -2388,21 +2367,10 @@ async def epoch_detail(epoch: int):
                 indent=2, sort_keys=True,
             )],
         ]
-    elif emission:
-        if legacy and len(emission.ranking or {}) <= 1:
-            ranking_nodes.append(["p.note",
-                "Single-contributor epoch — no LLM judgments."
-            ])
-        ranking_nodes.extend([
-            ["p", "Projected from Emission (no ranking Evidence event)."],
-            ["pre.blob", json.dumps(emission.ranking, indent=2, sort_keys=True)],
-        ])
+    elif ranking_started:
+        ranking_nodes = [["p.note", f"Ranking started: {ranking_started.event_id}"]]
     else:
-        ranking_nodes = [["p.note", "No ranking recorded."]]
-    if ranking_started and not ranking_completed:
-        ranking_nodes.insert(0, ["p.note",
-            f"Ranking started: {ranking_started.event_id}"
-        ])
+        ranking_nodes = [["p.note", "No ranking evidence."]]
 
     if emission:
         emission_node = _dl_rows([
@@ -2411,44 +2379,41 @@ async def epoch_detail(epoch: int):
             ("pool_after", emission.pool_after),
             ("discovery_snapshot_id", emission.discovery_snapshot_id),
             ("ranking_run_id", emission.ranking_run_id or None),
+            ("ranking_event_id", emission.ranking_event_id or None),
             ("models_used", ", ".join(emission.models_used or [])),
             ("distributions", json.dumps(emission.distributions, sort_keys=True)),
         ])
     else:
         emission_node = ["p.note", "No emission for this epoch."]
 
-    single_contributor = False
-    if discovery:
-        single_contributor = len({c.get("contributor") for c in discovery.commits}) <= 1
-    elif emission:
-        single_contributor = len(emission.ranking or {}) <= 1
+    contributors = {
+        e.payload.get("contributor")
+        for e in commit_evs
+        if e.payload.get("contributor")
+    }
+    no_comparisons_note = "No comparisons."
+    if len(contributors) <= 1:
+        no_comparisons_note += " Single-contributor — no LLM judgments."
 
     body = [
         _evidence_nav(),
         ["div.eyebrow", f"epoch {epoch}"],
         ["h1", f"epoch {epoch}"],
     ]
-    if legacy:
-        body.append(["p.note",
-            "Legacy epoch: projected from GitDiscovery/Emission without Evidence "
-            "envelopes. Eligible commits use discovery patches; discarded observation "
-            "metadata is marked legacy evidence unavailable. Single-contributor "
-            "epochs have no LLM judgments."
-        ])
-    if discovery:
+    if discovery_ev:
         body.extend([
             ["h2", "discovery"],
             _dl_rows([
-                ("snapshot_id", discovery.snapshot_id),
-                ("config_digest", discovery.config_digest),
-                ("initial_snapshot", discovery.initial_snapshot),
-                ("observations", len(discovery.observations)),
-                ("eligible", len(discovery.commits)),
+                ("snapshot_id", discovery_ev.payload.get("snapshot_id")),
+                ("config_digest", discovery_ev.payload.get("config_digest")),
+                ("observations", discovery_ev.payload.get("observation_count")),
+                ("eligible", discovery_ev.payload.get("eligible_count")),
+                ("event", _a(
+                    _evidence_path("event", discovery_ev.event_id),
+                    discovery_ev.event_id,
+                )),
             ]),
         ])
-    no_comparisons_note = "No comparisons."
-    if single_contributor:
-        no_comparisons_note += " Single-contributor — no LLM judgments."
     body.extend([
         ["h2", "commits"],
         _link_list(commit_links),
@@ -2471,92 +2436,42 @@ async def epoch_detail(epoch: int):
 @app.get("/commits/{commit_id}")
 async def commit_detail(commit_id: str):
     ev = find_evidence_payload("git.commit", "commit_id", commit_id)
-    discovery, legacy_row = (None, None)
     if not ev:
-        discovery, legacy_row = _legacy_commit_row(commit_id)
-    if not ev and not legacy_row:
-        discovery, obs = _legacy_observation(commit_id)
-        if obs is not None:
-            epoch = discovery.epoch if discovery else "?"
-            return _evidence_page(f"commit {commit_id[:24]}", [
-                _evidence_nav(
-                    _a(_evidence_path("epoch", str(epoch)), f"epoch {epoch}")
-                ),
-                ["div.eyebrow", "commit"],
-                ["h1", commit_id],
-                ["p.note", "legacy evidence unavailable"],
-                _dl_rows([
-                    ("oid", obs.get("oid")),
-                    ("eligible", obs.get("eligible")),
-                    ("exclusion_reason", obs.get("exclusion_reason")),
-                    ("epoch", str(epoch)),
-                ]),
-            ])
         return _evidence_page("commit not found", [
             _evidence_nav(),
             ["h1", "commit not found"],
             ["p", commit_id],
         ])
-    if ev:
-        p = ev.payload
-        epoch = ev.epoch
-        oid = p.get("oid", "")
-        contributor = p.get("contributor", "")
-        message = _blob_text(p.get("message"))
-        patch = _blob_text(p.get("patch"))
-        meta = _dl_rows([
+    p = ev.payload
+    epoch = ev.epoch
+    return _evidence_page(f"commit {commit_id[:24]}", [
+        _evidence_nav(_a(_evidence_path("epoch", str(epoch)), f"epoch {epoch}")),
+        ["div.eyebrow", "commit"],
+        ["h1", commit_id],
+        _dl_rows([
             ("commit_id", commit

… preview truncated; 7,279 characters omitted

download full diff B

Hardlinks — judgments / attempts / prompt

prompt download

judgments

attempts

Prompt text is loaded only by the download route.