You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [6d04afc2] refactor Side A — unified diff (full patch): diff --git a/Cargo.lock b/Cargo.lock index 2cea973082716e761ef6f5dd5886acc08ff9aac0..8c43fb75c472b102e6e1d3b837dce3355be898f2 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -17,6 +17,28 @@ version = "1.0.102" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7f202df86484c868dbad7eaa557ef785d5c66295e41b460ef922eca0723b842c" +[[package]] +name = "async-stream" +version = "0.3.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b5a71a6f37880a80d1d7f19efd781e4b5de42c88f0722cc13bcb6cc2cfe8476" +dependencies = [ + "async-stream-impl", + "futures-core", + "pin-project-lite", +] + +[[package]] +name = "async-stream-impl" +version = "0.3.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c7c24de15d275a1ecfd47a380fb4d5ec9bfe0933f309ed5e705b775596a3574d" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + [[package]] name = "async-trait" version = "0.1.89" @@ -1242,9 +1264,11 @@ dependencies = [ name = "sorter2-server" version = "0.0.1" dependencies = [ + "async-stream", "axum", "axum-extra", "dotenvy", + "futures-util", "maud", "reqwest", "serde", diff --git a/server/Cargo.toml b/server/Cargo.toml index bd600138b613bd0f546bdec217a5334cdcb20aa5..c940acb687fb141d21760a3d6656172013cf6f41 100644 --- a/server/Cargo.toml +++ b/server/Cargo.toml @@ -18,6 +18,8 @@ tracing = "0.1" tracing-subscriber = { version = "0.3", features = ["env-filter"] } reqwest = { version = "0.12", features = ["json"] } dotenvy = "0.15" +async-stream = "0.3" +futures-util = { version = "0.3", default-features = false, features = ["std"] } [dev-dependencies] reqwest = { version = "0.12", features = ["json"] } diff --git a/server/src/api/ui_html.rs b/server/src/api/ui_html.rs index b33a84e8bb5e817b26592868d88090e6d664d950..7af6527d03c483f33f3469ce6766c01a554c5fe3 100644 --- a/server/src/api/ui_html.rs +++ b/server/src/api/ui_html.rs @@ -6,7 +6,8 @@ use axum::{ use std::collections::HashMap; use crate::{ - html::{entity_section, input_panel, js_string_literal, ranking_panel, JsBuilder}, + fetch, + html::{input_panel, js_string_literal, ranking_panel, JsBuilder}, parser::parse_reddit_url, path_types::ItemId, reddit::ensure_partial_tree, @@ -89,18 +90,8 @@ pub async fn post_ui_html( }, HtmlUiAction::FetchEntity { item } => { let id = parse_item_param(&item); - if id.is_root() { - return ui_js_warn("nothing to fetch for the root").into_response(); - } - state.queue_entity_fetch(id.clone()); - let tree = state.tree.read().await; - let empty = crate::reducer::NodeState::default(); - let node = tree.get(&id).unwrap_or(&empty); - let panel = entity_section(&id, node, true); - JsBuilder::new() - .morph_selector("#entity-section", panel) - .into_response() - }, + fetch::fetch_entity_stream(state, id).into_response() + } } } diff --git a/server/src/fetch/html.rs b/server/src/fetch/html.rs new file mode 100644 index 0000000000000000000000000000000000000000..63634508496e224c38b9ec0308b7a6086462f925 --- /dev/null +++ b/server/src/fetch/html.rs @@ -0,0 +1,67 @@ +//! Markup for entity import / “Fetch from Reddit” (`POST /ui`, SSE response). + +use maud::{html, Markup}; + +use crate::{ + form_template::template_json_compact, + path_types::ItemId, + reddit::is_fetchable, + reducer::NodeState, + ui_action::UI_RPC_FIELD, +}; + +fn entity_panel(node: &NodeState) -> Markup { + html! { + @if let Some(data) = &node.data { + div id="entity-panel" class="entity-card" { + h2 { (data.title) } + @if let Some(author) = &data.author { + p class="muted small" { "by " (author) } + } + @if let Some(body) = &data.body_html { + div class="entity-body" { (maud::PreEscaped(body)) } + } + } + } + } +} + +/// Reddit/API import — `POST /ui` with `fetch_entity` returns an SSE stream. +pub fn fetch_entity_panel(item: &ItemId, has_data: bool, fetching: bool) -> Markup { + if !is_fetchable(item) { + return html! {}; + } + let label = if fetching { + "Fetching…" + } else if has_data { + "Fetch more" + } else { + "Fetch from Reddit" + }; + let rpc = template_json_compact(&serde_json::json!({ + "action": "fetch_entity", + "item": item.as_str(), + })) + .expect("fetch_entity rpc template"); + html! { + form method="post" action="/ui" id="fetch-entity-form" class="fetch-entity-form" { + input type="hidden" name=(UI_RPC_FIELD) value=(rpc); + @if fetching { + button type="submit" class="btn-secondary" disabled { (label) } + } @else { + button type="submit" class="btn-secondary" { (label) } + } + } + } +} + +/// Entity card + fetch control (target `#entity-section` for Idiomorph / SSE). +pub fn entity_section(item: &ItemId, node: &NodeState, fetching: bool) -> Markup { + let has_data = node.data.is_some(); + html! { + section id="entity-section" class="demo-panel" { + (entity_panel(node)) + (fetch_entity_panel(item, has_data, fetching)) + } + } +} diff --git a/server/src/fetch/mod.rs b/server/src/fetch/mod.rs new file mode 100644 index 0000000000000000000000000000000000000000..2290f9d3a0f1cbf1806c6339f82a4515c11cc3d3 --- /dev/null +++ b/server/src/fetch/mod.rs @@ -0,0 +1,115 @@ +//! Entity import over `POST /ui` as SSE (Reddit worker in [`crate::reddit`]). + +pub mod html; + +use std::convert::Infallible; +use std::time::Duration; + +use async_stream::stream; +use axum::response::sse::{Event, KeepAlive, Sse}; +use futures_util::Stream; +use serde::Serialize; +use tokio::sync::oneshot; + +use crate::{ + path_types::ItemId, + reddit::FetchJobResult, + reducer::NodeState, + state::AppState, +}; + +pub fn now_ms() -> i64 { + let t = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap_or_default(); + t.as_millis() as i64 +} + +#[derive(Serialize)] +struct SseMorphPayload { + selector: &'static str, + html: String, +} + +fn morph_complete_event(html: maud::Markup) -> Event { + let payload = SseMorphPayload { + selector: "#entity-section", + html: html.into_string(), + }; + let data = serde_json::to_string(&payload).unwrap_or_else(|_| "{}".into()); + Event::default().event("complete").data(data) +} + +/// Stream `fetching` → `complete` / `error` for [`crate::ui_action::HtmlUiAction::FetchEntity`]. +pub fn fetch_entity_stream( + state: AppState, + id: ItemId, +) -> Sse>> { + tracing::debug!(item = %id, "fetch entity stream opened"); + + let stream = stream! { + if id.is_root() { + yield Ok(Event::default().event("error").data("{\"message\":\"nothing to fetch for the root\"}")); + return; + } + + if !crate::reddit::is_fetchable(&id) { + tracing::debug!(item = %id, "fetch stream: not fetchable"); + yield Ok(Event::default().event("error").data("{\"message\":\"this page cannot be fetched from Reddit\"}")); + return; + } + + let fetching_html = { + let tree = state.tree.read().await; + let empty = NodeState::default(); + let node = tree.get(&id).unwrap_or(&empty); + html::entity_section(&id, node, true).into_string() + }; + let fetching_payload = serde_json::json!({ + "selector": "#entity-section", + "html": fetching_html, + }); + yield Ok(Event::default().event("fetching").data(fetching_payload.to_string())); + + let (tx, rx) = oneshot::channel(); + state.reddit.request_fetch(id.clone(), true, Some(tx)); + tracing::debug!(item = %id, "fetch stream: queued reddit job"); + + let result = match rx.await { + Ok(r) => r, + Err(_) => { + tracing::warn!(item = %id, "fetch stream: worker dropped oneshot"); + FetchJobResult::Failed("reddit worker stopped".into()) + } + }; + + tracing::debug!(item = %id, ?result, "fetch stream: job finished"); + + match result { + FetchJobResult::Imported | FetchJobResult::NotFound => { + let tree = state.tree.read().await; + let empty = NodeState::default(); + let node = tree.get(&id).unwrap_or(&empty); + yield Ok(morph_complete_event(html::entity_section(&id, node, false))); + } + FetchJobResult::SkippedCached | FetchJobResult::SkippedDuplicate => { + let tree = state.tree.read().await; + let empty = NodeState::default(); + let node = tree.get(&id).unwrap_or(&empty); + yield Ok(morph_complete_event(html::entity_section(&id, node, false))); + } + FetchJobResult::RateLimited { reset_secs } => { + yield Ok(Event::default().event("error").data( + serde_json::json!({"message": format!("Reddit rate limit — retry in {reset_secs}s")}).to_string(), + )); + } + FetchJobResult::Failed(msg) => { + yield Ok(Event::default().event("error").data( + serde_json::json!({"message": msg}).to_string(), + )); + } + } + }; + + Sse::new(stream).keep_alive(KeepAlive::new().interval(Duration::from_secs(15))) +} diff --git a/server/src/html/mod.rs b/server/src/html/mod.rs index db5b4c7f06b0be64603981166835cde268234f67..9314a7556306ddab969b896dbf4126b542a46722 100644 --- a/server/src/html/mod.rs +++ b/server/src/html/mod.rs @@ -7,10 +7,10 @@ use axum::{ use maud::{html, Markup, DOCTYPE}; use crate::{ + fetch::html::entity_section, form_template::template_json_compact, path_types::ItemId, ranking::{top_bottom, RankedItem}, - reddit::is_fetchable, reducer::{GroupState, NodeState}, state::AppState, ui_action::UI_RPC_FIELD, @@ -149,62 +149,6 @@ pub fn breadcrumb_path(item: &ItemId) -> Markup { } } -fn entity_panel(node: &NodeState) -> Markup { - html! { - @if let Some(data) = &node.data { - div id="entity-panel" class="entity-card" { - h2 { (data.title) } - @if let Some(author) = &data.author { - p class="muted small" { "by " (author) } - } - @if let Some(body) = &data.body_html { - div class="entity-body" { (maud::PreEscaped(body)) } - } - } - } - } -} - -/// Reddit/API import control — only shown on fetchable pages; never auto-fires. -pub fn fetch_entity_panel(item: &ItemId, has_data: bool, fetching: bool) -> Markup { - if !is_fetchable(item) { - return html! {}; - } - let label = if fetching { - "Fetching…" - } else if has_data { - "Fetch more" - } else { - "Fetch from Reddit" - }; - let rpc = template_json_compact(&serde_json::json!({ - "action": "fetch_entity", - "item": item.as_str(), - })) - .expect("fetch_entity rpc template"); - html! { - form method="post" action="/ui" id="fetch-entity-form" class="fetch-entity-form" { - input type="hidden" name=(UI_RPC_FIELD) value=(rpc); - @if fetching { - button type="submit" class="btn-secondary" disabled { (label) } - } @else { - button type="submit" class="btn-secondary" { (label) } - } - } - } -} - -/// Entity card + explicit fetch control (morphed as `#entity-section`). -pub fn entity_section(item: &ItemId, node: &NodeState, fetching: bool) -> Markup { - let has_data = node.data.is_some(); - html! { - section id="entity-section" class="demo-panel" { - (entity_panel(node)) - (fetch_entity_panel(item, has_data, fetching)) - } - } -} - fn rank_list(label: &str, items: &[RankedItem], start_rank: usize) -> Markup { html! { @if !items.is_empty() { diff --git a/server/src/lib.rs b/server/src/lib.rs index 79b173f391a96ae5d0d96fd656e1e8d2dd070d09..0677363e0ed21d244bfa065f7ec728549ddd9e50 100644 --- a/server/src/lib.rs +++ b/server/src/lib.rs @@ -1,6 +1,7 @@ pub mod api; pub mod event_log; pub mod events; +pub mod fetch; pub mod form_template; pub mod html; pub mod parser; diff --git a/server/src/reddit.rs b/server/src/reddit.rs index ff0f01e57b18af878eb5be3efc47204a7673589d..0e6ce32720951a58456852b67fb76a85dd9f4aad 100644 --- a/server/src/reddit.rs +++ b/server/src/reddit.rs @@ -7,12 +7,12 @@ use std::time::{Duration, Instant}; use reqwest::{header, Client, StatusCode}; use serde::Deserialize; use serde_json::Value; -use tokio::sync::{mpsc, RwLock}; +use tokio::sync::{mpsc, oneshot, RwLock}; use crate::{ event_log::EventLog, events::Event, - html::now_ms, + fetch::now_ms, path_types::ItemId, reducer::GlobalTree, }; @@ -22,10 +22,21 @@ pub fn ensure_partial_tree(tree: &mut GlobalTree, id: &ItemId) { tree.ensure_path(id); } +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum FetchJobResult { + Imported, + NotFound, + SkippedDuplicate, + SkippedCached, + RateLimited { reset_secs: u64 }, + Failed(String), +} + pub struct RedditCommand { pub id: ItemId, /// User-initiated fetch bypasses the in-memory "recently fetched" cache. pub force: bool, + pub done: Option>, } #[derive(Clone)] @@ -72,14 +83,29 @@ impl RedditBroker { .build() .expect("reqwest client"); + tracing::debug!( + api_base = %config.api_base, + oauth_base = %config.oauth_base, + oauth = config.creds.is_some(), + "reddit worker started" + ); + tokio::spawn(reddit_worker(rx, tree, event_log, client, config)); Self { tx } } /// Queue a fetch; drops when the channel is full (backpressure). - pub fn request_fetch(&self, id: ItemId, force: bool) { - let _ = self.tx.try_send(RedditCommand { id, force }); + pub fn request_fetch( + &self, + id: ItemId, + force: bool, + done: Option>, + ) { + match self.tx.try_send(RedditCommand { id: id.clone(), force, done }) { + Ok(()) => tracing::debug!(item = %id, force, "reddit fetch queued"), + Err(_) => tracing::warn!(item = %id, "reddit fetch queue full, dropped"), + } } } @@ -95,8 +121,6 @@ impl RedditApiConfig { } impl RedditCredentials { - /// Reddit's OAuth docs call these "client id" and "client secret"; the app - /// registration UI often labels them "app id" / "app secret" — same values. fn from_env() -> Option { let client_id = std::env::var("REDDIT_CLIENT_ID") .or_else(|_| std::env::var("REDDIT_APP_ID")) @@ -128,12 +152,10 @@ pub fn default_user_agent() -> String { }) } -/// True when this node can be loaded from the Reddit JSON API. pub fn is_fetchable(id: &ItemId) -> bool { !map_item_to_reddit_api(id, "https://example.com").is_empty() } -/// Derive UI-facing fields from a stored payload (Reddit-specific when under reddit.com). pub fn entity_view_from_payload(id: &ItemId, payload: &Value) -> Option { if id.as_str().starts_with("reddit.com") { return parse_reddit_view(id, payload); @@ -141,12 +163,17 @@ pub fn entity_view_from_payload(id: &ItemId, payload: &Value) -> Option>, result: FetchJobResult) { + if let Some(tx) = done { + let _ = tx.send(result); + } +} + async fn reddit_worker( mut rx: mpsc::Receiver, tree: Arc>, @@ -168,15 +195,25 @@ async fn reddit_worker( recently_fetched.retain(|_, t| now.duration_since(*t) < cache_ttl); if in_flight.contains(&cmd.id) { + tracing::debug!(item = %cmd.id, "reddit fetch skipped: already in flight"); + notify(cmd.done, FetchJobResult::SkippedDuplicate); continue; } if !cmd.force && recently_fetched.contains_key(&cmd.id) { + tracing::debug!(item = %cmd.id, "reddit fetch skipped: recently fetched cache"); + notify(cmd.done, FetchJobResult::SkippedCached); continue; } in_flight.insert(cmd.id.clone()); let fetch_id = cmd.id.clone(); + let done = cmd.done; + tracing::debug!( + item = %fetch_id, + delay_ms = current_delay.as_millis(), + "reddit fetch starting after delay" + ); tokio::time::sleep(current_delay).await; if let Some(c) = &creds { @@ -184,8 +221,18 @@ async fn reddit_worker( } let token = oauth.as_ref().map(|t| t.access_token.as_str()); + if token.is_some() { + tracing::debug!(item = %fetch_id, "reddit fetch using OAuth bearer"); + } + + let fetch_base = if token.is_some() { + &oauth_base + } else { + &api_base + }; + let outcome = do_fetch(&client, fetch_base, &fetch_id, token).await; - match do_fetch(&client, &api_base, &fetch_id, token).await { + match outcome { Ok(FetchOutcome::Payload(payload)) => { let ts = now_ms(); let event = Event::EntityImported { @@ -193,30 +240,49 @@ async fn reddit_worker( ts, payload: payload.clone(), }; - if let Err(e) = event_log.append(&event).await { - tracing::warn!("event log append failed for {}: {}", fetch_id, e); - } else { - apply_entity_import(&mut *tree.write().await, &fetch_id, payload); - recently_fetched.insert(fetch_id.clone(), Instant::now()); - current_delay = Duration::from_millis(600); + tracing::debug!( + item = %fetch_id, + ts, + payload_keys = ?payload.as_object().map(|o| o.len()), + "reddit fetch got JSON payload, appending event" + ); + match event_log.append(&event).await { + Err(e) => { + tracing::warn!( + item = %fetch_id, + err = %e, + "reddit event log append failed" + ); + notify(done, FetchJobResult::Failed(e.to_string())); + } + Ok(()) => { + apply_entity_import(&mut *tree.write().await, &fetch_id, payload); + recently_fetched.insert(fetch_id.clone(), Instant::now()); + current_delay = Duration::from_millis(600); + tracing::info!(item = %fetch_id, "reddit entity imported"); + notify(done, FetchJobResult::Imported); + } } } Ok(FetchOutcome::NotFound) => { + tracing::debug!(item = %fetch_id, "reddit fetch: not found (no event written)"); recently_fetched.insert(fetch_id.clone(), Instant::now()); + notify(done, FetchJobResult::NotFound); } Ok(FetchOutcome::RateLimited { reset_secs }) => { - let wait = Duration::from_secs(reset_secs.max(1)); tracing::warn!( - "Reddit rate limit for {}; sleeping {}s", - fetch_id, - wait.as_secs() + item = %fetch_id, + reset_secs, + "reddit rate limited" ); - tokio::time::sleep(wait).await; + tokio::time::sleep(Duration::from_secs(reset_secs.max(1))).await; current_delay = (current_delay * 2).min(Duration::from_secs(60)); + notify(done, FetchJobResult::RateLimited { reset_secs }); } Err(e) => { - tracing::warn!("Reddit fetch failed for {}: {}", fetch_id, e); + tracing::warn!(item = %fetch_id, err = %e, "reddit fetch failed"); current_delay = (current_delay * 2).min(Duration::from_secs(60)); + notify(done, FetchJobResult::Failed(e)); } } @@ -238,6 +304,7 @@ async fn ensure_oauth_token( ) -> Option { if let Some(t) = existing { if Instant::now() < t.expires_at - Duration::from_secs(60) { + tracing::debug!("reddit OAuth token still valid"); return Some(t); } } @@ -246,6 +313,7 @@ async fn ensure_oauth_token( "{}/api/v1/access_token", oauth_base.trim_end_matches('/') ); + tracing::debug!(%url, "reddit OAuth token request"); let resp = client .post(&url) @@ -257,13 +325,13 @@ async fn ensure_oauth_token( let resp = match resp { Ok(r) => r, Err(e) => { - tracing::warn!("Reddit OAuth token request failed: {e}"); + tracing::warn!("reddit OAuth token request failed: {e}"); return None; } }; if !resp.status().is_success() { - tracing::warn!("Reddit OAuth token HTTP {}", resp.status()); + tracing::warn!("reddit OAuth token HTTP {}", resp.status()); return None; } @@ -276,11 +344,12 @@ async fn ensure_oauth_token( let body: TokenResponse = match resp.json().await { Ok(b) => b, Err(e) => { - tracing::warn!("Reddit OAuth token parse failed: {e}"); + tracing::warn!("reddit OAuth token parse failed: {e}"); return None; } }; + tracing::debug!(expires_in = body.expires_in, "reddit OAuth token acquired"); Some(OAuthToken { access_token: body.access_token, expires_at: Instant::now() + Duration::from_secs(body.expires_in), @@ -295,35 +364,72 @@ async fn do_fetch( ) -> Result { let url = map_item_to_reddit_api(id, api_base); if url.is_empty() { + tracing::debug!(item = %id, "reddit do_fetch: no API URL for item"); return Ok(FetchOutcome::NotFound); } + tracing::debug!(item = %id, %url, bearer = bearer.is_some(), "reddit HTTP GET"); + let mut req = client.get(&url); if let Some(token) = bearer { req = req.bearer_auth(token); } - let resp = req.send().await.map_err(|e| e.to_string())?; + let resp = req.send().await.map_err(|e| { + tracing::debug!(item = %id, %url, err = %e, "reddit HTTP transport error"); + e.to_string() + })?; + + let status = resp.status(); + tracing::debug!( + item = %id, + %url, + %status, + remaining = ?rate_limit_remaining(&resp), + reset = ?rate_limit_reset_secs(&resp), + "reddit HTTP response" + ); - if resp.status() == StatusCode::TOO_MANY_REQUESTS { + if status == StatusCode::TOO_MANY_REQUESTS { let reset = rate_limit_reset_secs(&resp); return Ok(FetchOutcome::RateLimited { reset_secs: reset }); } - if resp.status() == StatusCode::SERVICE_UNAVAILABLE { - return Err("Reddit unavailable (503)".to_string()); + if status == StatusCode::SERVICE_UNAVAILABLE { + return Err("Reddit unavailable (503)".into()); } - if !resp.status().is_success() { + if !status.is_success() { + let body = resp.text().await.unwrap_or_default(); + tracing::debug!( + item = %id, + %status, + body_len = body.len(), + body_prefix = %body.chars().take(240).collect::(), + "reddit non-success body" + ); return Ok(FetchOutcome::NotFound); } if rate_limit_remaining(&resp) == Some(0) { let reset = rate_limit_reset_secs(&resp); + tracing::debug!(item = %id, reset_secs = reset, "reddit headers: rate limit exhausted"); return Ok(FetchOutcome::RateLimited { reset_secs: reset }); } - let payload: Value = resp.json().await.map_err(|e| e.to_string())?; + let text = resp.text().await.map_err(|e| e.to_string())?; + tracing::debug!(item = %id, bytes = text.len(), "reddit response body received"); + + let payload: Value = serde_json::from_str(&text).map_err(|e| { + tracing::debug!( + item = %id, + err = %e, + body_prefix = %text.chars().take(240).collect::(), + "reddit JSON parse failed" + ); + format!("invalid JSON: {e}") + })?; + Ok(FetchOutcome::Payload(payload)) } @@ -344,7 +450,6 @@ fn rate_limit_reset_secs(resp: &reqwest::Response) -> u64 { .unwrap_or(5) } -/// Map canonical item id to a Reddit JSON API URL under `api_base`. pub fn map_item_to_reddit_api(id: &ItemId, api_base: &str) -> String { let path = id.as_str(); if !path.starts_with("reddit.com/") && path != "reddit.com" { @@ -445,26 +550,6 @@ mod tests { map_item_to_reddit_api(&id, "https://www.reddit.com"), "https://www.reddit.com/r/rust/about.json?raw_json=1" ); - assert_eq!( - map_item_to_reddit_api(&id, "http://127.0.0.1:9999"), - "http://127.0.0.1:9999/r/rust/about.json?raw_json=1" - ); - } - - #[test] - fn map_post_url() { - let id = ItemId::parse("reddit.com/r/amitheasshole/comments/1trnvdl").unwrap(); - assert_eq!( - map_item_to_reddit_api(&id, "https://www.reddit.com"), - "https://www.reddit.com/r/amitheasshole/comments/1trnvdl.json?raw_json=1" - ); - } - - #[test] - fn is_fetchable_reddit_sub() { - let id = ItemId::parse("reddit.com/r/rust").unwrap(); - assert!(is_fetchable(&id)); - assert!(!is_fetchable(&ItemId::opaque("example.com/x"))); } #[test] @@ -477,19 +562,5 @@ mod tests { ) .unwrap(); assert_eq!(entity.title, "The Rust Programming Language"); - assert!(entity.body_html.as_ref().is_some_and(|b| b.contains("Rust"))); - } - - #[test] - fn parse_post_fixture() { - let json = r#"[{"kind":"Listing","data":{"children":[{"kind":"t3","data":{"title":"AITA","author":"op","selftext_html":"<p>hi</p>","thumbnail":"https://b.thumbs.redditmedia.com/x.jpg"}}]}}]"#; - let v: Value = serde_json::from_str(json).unwrap(); - let entity = entity_view_from_payload( - &ItemId::parse("reddit.com/r/x/comments/abc").unwrap(), - &v, - ) - .unwrap(); - assert_eq!(entity.title, "AITA"); - assert_eq!(entity.author.as_deref(), Some("op")); } } diff --git a/server/src/state.rs b/server/src/state.rs index d71d1079a8f486cdab15795384aef0b81b32544d..e7ff9f663e45b5e168d6a3869948e3bd890966d4 100644 --- a/server/src/state.rs +++ b/server/src/state.rs @@ -143,9 +143,9 @@ impl AppState { Ok(()) } - /// User-initiated Reddit/API import (via "Fetch more" — never on paste or navigate). - pub fn queue_entity_fetch(&self, id: ItemId) { - self.reddit.request_fetch(id, true); + /// User-initiated Reddit/API import (SSE / fetch module only). + pub fn queue_entity_fetch(&self, id: ItemId, done: Option>) { + self.reddit.request_fetch(id, true, done); } pub async fn record_vote( diff --git a/server/src/ui_action.rs b/server/src/ui_action.rs index 3d6a49a2a3fb950752827efe1f5a049308f09baf..ef9ac873fc0442e0b960f144309c8e91124d7884 100644 --- a/server/src/ui_action.rs +++ b/server/src/ui_action.rs @@ -26,7 +26,7 @@ pub enum HtmlUiAction { ParseQuery { query: String, }, - /// Fetch upstream entity data for the current page (explicit user action only). + /// Import entity data; `POST /ui` responds with `text/event-stream` (not JS). FetchEntity { item: String, }, diff --git a/server/static/sorter_ui.js b/server/static/sorter_ui.js index d9b016f197547f37ca4b2bcdd7ee6b673d0fb3f0..fe7bfc657c0fccae5c596d005b0e807abefec677 100644 --- a/server/static/sorter_ui.js +++ b/server/static/sorter_ui.js @@ -1,5 +1,5 @@ /** - * sorter2 web UI: fetch/eval for POST /ui. No product logic here. + * sorter2 web UI: POST /ui returns JS (morph) or SSE (entity fetch). */ (function () { function evalJs(js) { @@ -8,15 +8,98 @@ } } + function morphSelector(selector, html) { + var el = document.querySelector(selector); + if (el && typeof Idiomorph !== 'undefined') { + Idiomorph.morph(el, html); + } + } + + function handleSseEvent(eventType, data, form) { + if (eventType === 'fetching' || eventType === 'complete') { + try { + var msg = JSON.parse(data); + morphSelector(msg.selector || '#entity-section', msg.html); + } catch (err) { + console.warn('fetch morph parse', err); + } + } + if (eventType === 'complete' || eventType === 'error') { + var btn = form && form.querySelector('button[type="submit"]'); + if (btn) btn.disabled = false; + } + if (eventType === 'error') { + try { + var err = JSON.parse(data); + console.warn('fetch error:', err.message || data); + } catch (_e) { + console.warn('fetch error:', data); + } + } + } + + function consumeSseStream(response, form) { + var reader = response.body.getReader(); + var decoder = new TextDecoder(); + var buffer = ''; + var eventType = ''; + var dataLines = []; + + function dispatch() { + if (!eventType && dataLines.length === 0) return; + handleSseEvent(eventType || 'message', dataLines.join('\n'), form); + eventType = ''; + dataLines = []; + } + + function pump() { + return reader.read().then(function (chunk) { + if (chunk.done) { + dispatch(); + return; + } + buffer += decoder.decode(chunk.value, { stream: true }); + var parts = buffer.split('\n'); + buffer = parts.pop() || ''; + for (var i = 0; i < parts.length; i++) { + var line = parts[i].replace(/\r$/, ''); + if (line === '') { + dispatch(); + } else if (line.indexOf('event:') === 0) { + eventType = line.slice(6).trim(); + } else if (line.indexOf('data:') === 0) { + dataLines.push(line.slice(5).trim()); + } + } + return pump(); + }); + } + + return pump(); + } + function postUiForm(form) { + var btn = form.querySelector('button[type="submit"]'); + if (form.id === 'fetch-entity-form' && btn) { + btn.disabled = true; + } return fetch(form.action, { method: 'POST', body: new URLSearchParams(new FormData(form)), headers: { 'Content-Type': 'application/x-www-form-urlencoded' }, credentials: 'same-origin', }).then(function (resp) { - return resp.text(); - }).then(evalJs); + var ct = resp.headers.get('content-type') || ''; + if (ct.indexOf('text/event-stream') !== -1) { + return consumeSseStream(resp, form); + } + return resp.text().then(evalJs); + }).catch(function (err) { + if (form.id === 'fetch-entity-form' && btn) { + btn.disabled = false; + } + console.warn('POST /ui failed', err); + }); } function initSorterUi() { diff --git a/test/reddit_import.clj b/test/reddit_import.clj index 54edaedeef08718c8f184d6aa468d8e6415068c7..17f2ee39e1498b0b6bb6a9d8a7e35608b0bd1b32 100644 --- a/test/reddit_import.clj +++ b/test/reddit_import.clj @@ -44,10 +44,12 @@ (do (Thread/sleep 200) (recur)) false)))))) -(defn- curl-post-ui [base rpc-json] +(defn- curl-fetch-ui-sse [base item] (process/shell {:out :string :err :string} - "curl" "-sf" "-X" "POST" (str base "/ui") - "--data-urlencode" (str "__rpc__=" rpc-json))) + "curl" "-sfN" "--max-time" "20" + "-X" "POST" (str base "/ui") + "--data-urlencode" + (str "__rpc__={\"action\":\"fetch_entity\",\"item\":\"" item "\"}"))) (defn- wait-event-log [path ms] (let [deadline (+ (System/currentTimeMillis) ms)] @@ -98,11 +100,12 @@ "curl" "-sf" browse-url))] (is (str/includes? before "Fetch from Reddit")) (is (not (str/includes? before "The Rust Programming Language"))) - (let [rpc "{\"action\":\"fetch_entity\",\"item\":\"reddit.com/r/rust\"}" - post (curl-post-ui app-base rpc) - log-path (str data-dir "/events.jsonl")] - (is (zero? (:exit post)) "fetch_entity POST succeeds") - (is (wait-event-log log-path 10000) "event log written") + (let [log-path (str data-dir "/events.jsonl") + sse (curl-fetch-ui-sse app-base "reddit.com/r/rust")] + (is (zero? (:exit sse)) "POST /ui fetch_entity SSE succeeds") + (is (str/includes? (:out sse) "event: complete")) + (is (str/includes? (:out sse) "The Rust Programming Language")) + (is (wait-event-log log-path 2000) "event log written") (let [after (:out (process/shell {:out :string :err :string} "curl" "-sf" browse-url)) log (slurp (io/file log-path))] Side B — contributor: tommy-mor Side B — commit message: [c7ef287e] Rank every eligible commit with the LLM council. Stop short-circuiting on a single contributor; pairwise-sort commits, roll scores up for emission payouts, and surface commit rankings on epoch pages. Co-authored-by: Cursor Side B — unified diff (full patch): diff --git a/constitution.py b/constitution.py index 26ba130e885e0e69fb7874ca5c3f07f42100a150..71fc46b4c9a4d7a9ea7bf319860b0db1acc4a673 100644 --- a/constitution.py +++ b/constitution.py @@ -503,20 +503,22 @@ def _epochs_in_ledger() -> list[int]: def build_pairwise_prompt(side_a: dict, side_b: dict) -> str: - return f"""You are ranking contributions to an open source project. -Compare these two sides (each may be one or more commits). Decide which side contributed more. + return f"""You are ranking individual git commits to an open source project. +Compare these two commits. Decide which commit contributed more. Return ONLY a JSON object: {{"winner": "A" or "B", "ratio": "N:M", "explanation": "..."}} -Side A — commit messages: +Side A — contributor: {side_a.get('contributor', '?')} +Side A — commit message: {side_a['message']} -Side A — unified diffs (full patches): +Side A — unified diff (full patch): {side_a['diff']} -Side B — commit messages: +Side B — contributor: {side_b.get('contributor', '?')} +Side B — commit message: {side_b['message']} -Side B — unified diffs (full patches): +Side B — unified diff (full patch): {side_b['diff']}""" @@ -1518,16 +1520,28 @@ async def broadcast_js(js: str): await queue.put(js) -def _author_side_for_llm(author: str, author_commits: dict) -> dict: - cs = author_commits[author] +def _commit_side_for_llm(row: dict) -> dict: + oid = row["oid"] + short = oid.split(":", 1)[1][:8] if ":" in oid else oid[:8] return { - "message": "\n".join(f"[{c['sha']}] {c['message']}" for c in cs), - "diff": "\n\n".join(f"=== {c['sha']} ===\n{c['diff']}" for c in cs), - "commit_ids": [c["commit_id"] for c in cs], - "contributor": author, + "message": f"[{short}] {row['message']}", + "diff": row["patch"] or "", + "commit_id": commit_id_for_oid(oid), + "contributor": row["contributor"], + "oid": oid, } +def _rollup_contributor_scores( + ordered: list[dict], commit_scores: list[Decimal] +) -> dict[str, Decimal]: + totals: dict[str, Decimal] = {} + for row, score in zip(ordered, commit_scores): + contributor = row["contributor"] + totals[contributor] = totals.get(contributor, Decimal("0")) + score + return totals + + def _find_judgment(comparison_id: str, model_id: str) -> dict | None: for e in evidence_by_kind("llm.judgment"): p = e.payload @@ -1546,101 +1560,106 @@ def _find_ranking_models(ranking_run_id: str) -> list[str] | None: async def rank_commits(commits: list[dict], *, epoch: int = -1): + """Pairwise-rank every eligible commit; roll scores up to contributors.""" if not commits: return {}, [], {"ranking_run_id": "", "ranking_event_id": ""} - commit_ids = sorted(commit_id_for_oid(row["oid"]) for row in commits) + ordered = sorted(commits, key=lambda r: r["oid"]) + commit_ids = [commit_id_for_oid(row["oid"]) for row in ordered] ranking_run_id = _content_id("rank", { "epoch": epoch, - "commit_ids": commit_ids, + "commit_ids": sorted(commit_ids), }) - contributors = sorted(set(c["contributor"] for c in commits)) + contributors = sorted({c["contributor"] for c in ordered}) - if len(contributors) == 1: + # Nothing to compare: a single commit (not a single contributor). + if len(ordered) == 1: await append_evidence(epoch, "ranking.started", { "ranking_run_id": ranking_run_id, "commit_ids": commit_ids, "contributors": contributors, "models": [], - "summary": f"ranking epoch {epoch}: single contributor", + "summary": f"ranking epoch {epoch}: single commit", }) - ranking = {contributors[0]: Decimal("1")} + commit_ranking = {commit_ids[0]: "1"} + contributor_ranking = {ordered[0]["contributor"]: Decimal("1")} completed = await append_evidence(epoch, "ranking.completed", { "ranking_run_id": ranking_run_id, "models": [], - "ranking": {contributors[0]: "1"}, + "commit_ranking": commit_ranking, + "contributor_ranking": {ordered[0]["contributor"]: "1"}, + "ranking": {ordered[0]["contributor"]: "1"}, "judgment_ids": [], - "summary": f"Only {contributors[0]} is eligible; rank is 1.0", + "summary": f"Only one eligible commit; {ordered[0]['contributor']} rank 1.0", }) await broadcast_audit( "ranking", - f"Only {contributors[0]} is eligible; rank is 1.0", + f"Only one eligible commit; {ordered[0]['contributor']} rank 1.0", progress=90, phase="finalizing", evidence_event_id=completed.event_id, evidence_url=_evidence_url("event", completed.event_id), links={"epoch": _evidence_url("epoch", str(epoch))}, ) - return ranking, [], { + return contributor_ranking, [], { "ranking_run_id": ranking_run_id, "ranking_event_id": completed.event_id, } if not (OPENROUTER_API_KEY or "").strip(): raise RuntimeError( - "OPENROUTER_API_KEY is required when multiple contributors need ranking" + "OPENROUTER_API_KEY is required when multiple commits need ranking" ) models = _find_ranking_models(ranking_run_id) if models is None: models = await fetch_top_models(n=3) if not models: - raise RuntimeError("no council models available for contributor ranking") + raise RuntimeError("no council models available for commit ranking") await append_evidence(epoch, "ranking.started", { "ranking_run_id": ranking_run_id, "commit_ids": commit_ids, "contributors": contributors, "models": models, - "summary": f"Council selected: {', '.join(models)}", + "summary": ( + f"Council selected: {', '.join(models)} — " + f"{len(ordered)} commits" + ), }) await broadcast_audit( "council", - f"Council selected: {', '.join(models)}", + f"Council selected: {', '.join(models)} — ranking {len(ordered)} commits", progress=35, phase="ranking", ) await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ - ["div.log-council", f"Council: {', '.join(models)} — {len(commits)} commits"] + ["div.log-council", + f"Council: {', '.join(models)} — {len(ordered)} commits"] ])) - authors = contributors - author_commits = {a: [] for a in authors} - for row in sorted(commits, key=lambda r: r["oid"]): - author_commits[row["contributor"]].append({ - "message": row["message"], - "sha": row["oid"].split(":", 1)[1][:8], - "diff": row["patch"], - "commit_id": commit_id_for_oid(row["oid"]), - }) - + sides = [_commit_side_for_llm(row) for row in ordered] judgment_ids: list[str] = [] async def compare_fn(i, j): - a1, a2 = authors[i], authors[j] - side_a = _author_side_for_llm(a1, author_commits) - side_b = _author_side_for_llm(a2, author_commits) + side_a, side_b = sides[i], sides[j] + label_a = f"{side_a['commit_id'][:16]} ({side_a['contributor']})" + label_b = f"{side_b['commit_id'][:16]} ({side_b['contributor']})" prompt = build_pairwise_prompt(side_a, side_b) comparison_material = { "ranking_run_id": ranking_run_id, "side_a": { - "contributor": a1, - "commit_ids": side_a["commit_ids"], + "contributor": side_a["contributor"], + "commit_id": side_a["commit_id"], + "commit_ids": [side_a["commit_id"]], + "oid": side_a["oid"], "message": _bytes_blob(side_a["message"]), "diff": _bytes_blob(side_a["diff"]), }, "side_b": { - "contributor": a2, - "commit_ids": side_b["commit_ids"], + "contributor": side_b["contributor"], + "commit_id": side_b["commit_id"], + "commit_ids": [side_b["commit_id"]], + "oid": side_b["oid"], "message": _bytes_blob(side_b["message"]), "diff": _bytes_blob(side_b["diff"]), }, @@ -1650,22 +1669,24 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1): comparison_material = { **comparison_material, "comparison_id": comparison_id, - "summary": f"Comparing {a1} with {a2}", + "summary": f"Comparing {label_a} with {label_b}", } cmp_ev = await append_evidence(epoch, "comparison.input", comparison_material) await broadcast_audit( "comparison", - f"Comparing {a1} with {a2}", + f"Comparing commits {label_a} vs {label_b}", phase="ranking", evidence_event_id=cmp_ev.event_id, evidence_url=_evidence_url("comparison", comparison_id), links={ "comparison": _evidence_url("comparison", comparison_id), + "commit_a": _evidence_url("commit", side_a["commit_id"]), + "commit_b": _evidence_url("commit", side_b["commit_id"]), "epoch": _evidence_url("epoch", str(epoch)), }, ) await broadcast_js(exec_event(Three[Selector("#emission-status")][MORPH][ - ["div#emission-status", f"Comparing {a1} vs {a2}…"] + ["div#emission-status", f"Comparing {label_a} vs {label_b}…"] ])) results = [] for model in models: @@ -1697,9 +1718,15 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1): jud_id = (existing or _find_judgment(comparison_id, model) or {}).get( "judgment_id" ) + win_label = ( + f"{sides[w]['commit_id'][:16]} ({sides[w]['contributor']})" + ) + lose_label = ( + f"{sides[l]['commit_id'][:16]} ({sides[l]['contributor']})" + ) await broadcast_audit( "vote", - f"{model}: {authors[w]} over {authors[l]} ({result['ratio']})", + f"{model}: {win_label} over {lose_label} ({result['ratio']})", phase="ranking", evidence_url=( _evidence_url("judgment", jud_id) if jud_id else None @@ -1714,8 +1741,8 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1): await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ ["div.log-vote", ["span.model", model], " — ", - ["span.winner", authors[w]], f" beat ", - ["span.loser", authors[l]], f" ({result['ratio']}) ", + ["span.winner", win_label], f" beat ", + ["span.loser", lose_label], f" ({result['ratio']}) ", ["span.explanation", result["explanation"]], ] ])) @@ -1745,24 +1772,46 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1): ["div#emission-status", label] ])) - pairs = await pairwise_rank(len(authors), compare_fn, progress_fn) + pairs = await pairwise_rank(len(ordered), compare_fn, progress_fn) if not pairs: - ranking = {authors[0]: Decimal("1")} if authors else {} + commit_score_list = [Decimal("1")] else: scores = rank_centrality(pairs) - ranking = {authors[i]: Decimal(str(scores[i])) for i in range(len(authors))} - ranking_rows = sorted(ranking.items(), key=lambda x: x[1], reverse=True) + commit_score_list = [Decimal(str(scores[i])) for i in range(len(ordered))] + + commit_ranking = { + commit_ids[i]: str(commit_score_list[i]) for i in range(len(ordered)) + } + contributor_totals = _rollup_contributor_scores(ordered, commit_score_list) + contrib_rows = sorted( + contributor_totals.items(), key=lambda x: x[1], reverse=True + ) + commit_rows = sorted( + ((commit_ids[i], commit_score_list[i], ordered[i]["contributor"]) + for i in range(len(ordered))), + key=lambda x: x[1], + reverse=True, + ) completed = await append_evidence(epoch, "ranking.completed", { "ranking_run_id": ranking_run_id, "models": models, - "ranking": {a: str(s) for a, s in ranking_rows}, + "commit_ranking": commit_ranking, + "contributor_ranking": {a: str(s) for a, s in contrib_rows}, + "ranking": {a: str(s) for a, s in contrib_rows}, "judgment_ids": judgment_ids, - "summary": "Ranking: " + ", ".join(f"{a} {s:.4f}" for a, s in ranking_rows), + "summary": ( + "Commit ranking: " + + ", ".join( + f"{cid[:16]}={float(s):.4f}" for cid, s, _ in commit_rows[:12] + ) + + ("…" if len(commit_rows) > 12 else "") + ), }) await broadcast_audit( "ranking", - "Ranking: " + ", ".join(f"{a} {s:.4f}" for a, s in ranking_rows), + "Contributor rollup: " + + ", ".join(f"{a} {s:.4f}" for a, s in contrib_rows), progress=90, phase="finalizing", evidence_event_id=completed.event_id, @@ -1772,10 +1821,10 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1): await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ ["div.log-ranking", ["b", "Ranking: "], - *[["span.rank-entry", f"{a} {float(s):.3f} "] for a, s in ranking_rows], + *[["span.rank-entry", f"{a} {float(s):.3f} "] for a, s in contrib_rows], ] ])) - return ranking, models, { + return contributor_totals, models, { "ranking_run_id": ranking_run_id, "ranking_event_id": completed.event_id, } @@ -2352,6 +2401,23 @@ async def epoch_detail(epoch: int): ranking_nodes: list = [] if ranking_completed: + commit_ranking = ranking_completed.payload.get("commit_ranking") or {} + contrib_ranking = ( + ranking_completed.payload.get("contributor_ranking") + or ranking_completed.payload.get("ranking") + or {} + ) + commit_rank_links = [ + ( + f"{cid[:20]} = {score}", + _evidence_path("commit", cid), + ) + for cid, score in sorted( + commit_ranking.items(), + key=lambda kv: Decimal(str(kv[1])), + reverse=True, + ) + ] ranking_nodes = [ ["p", ranking_completed.payload.get("summary") or "ranking completed"], _dl_rows([ @@ -2362,10 +2428,10 @@ async def epoch_detail(epoch: int): ranking_completed.event_id, )), ]), - ["pre.blob", json.dumps( - ranking_completed.payload.get("ranking") or {}, - indent=2, sort_keys=True, - )], + ["h3", "commit ranking"], + _link_list(commit_rank_links) if commit_rank_links else ["p.note", "(none)"], + ["h3", "contributor rollup"], + ["pre.blob", json.dumps(contrib_ranking, indent=2, sort_keys=True)], ] elif ranking_started: ranking_nodes = [["p.note", f"Ranking started: {ranking_started.event_id}"]] @@ -2386,14 +2452,9 @@ async def epoch_detail(epoch: int): else: emission_node = ["p.note", "No emission for this epoch."] - contributors = { - e.payload.get("contributor") - for e in commit_evs - if e.payload.get("contributor") - } no_comparisons_note = "No comparisons." - if len(contributors) <= 1: - no_comparisons_note += " Single-contributor — no LLM judgments." + if len(commit_evs) <= 1: + no_comparisons_note += " Fewer than two eligible commits — nothing to pairwise-rank." body = [ _evidence_nav(), @@ -2511,9 +2572,14 @@ async def comparison_detail(comparison_id: str): j.payload.get("summary") or jid, _evidence_path("judgment", jid), )) - commit_links = [] - for cid in (side_a.get("commit_ids") or []) + (side_b.get("commit_ids") or []): - commit_links.append((cid, _evidence_path("commit", cid))) + commit_ids = [] + for side in (side_a, side_b): + cid = side.get("commit_id") + if cid: + commit_ids.append(cid) + else: + commit_ids.extend(side.get("commit_ids") or []) + commit_links = [(cid, _evidence_path("commit", cid)) for cid in commit_ids] return _evidence_page(f"comparison {comparison_id[:24]}", [ _evidence_nav(_a(_evidence_path("epoch", str(ev.epoch)), f"epoch {ev.epoch}")), ["div.eyebrow", "comparison"], @@ -2522,8 +2588,8 @@ async def comparison_detail(comparison_id: str): ("summary", p.get("summary")), ("ranking_run_id", p.get("ranking_run_id")), ("evidence_event", _a(_evidence_path("event", ev.event_id), ev.event_id)), - ("side_a", side_a.get("contributor")), - ("side_b", side_b.get("contributor")), + ("side_a", f"{side_a.get('commit_id', '?')} ({side_a.get('contributor', '?')})"), + ("side_b", f"{side_b.get('commit_id', '?')} ({side_b.get('contributor', '?')})"), ]), ["p", _a(f"/comparisons/{comparison_id}/prompt", "download prompt")], ["h2", "commits"], @@ -2534,12 +2600,12 @@ async def comparison_detail(comparison_id: str): _link_list(judgment_links), ["h2", "prompt"], _pre_blob(_blob_text(p.get("prompt"))), - ["h2", f"side A — {side_a.get('contributor', '?')}"], + ["h2", f"side A — {side_a.get('commit_id', side_a.get('contributor', '?'))}"], ["h3", "message"], _pre_blob(_blob_text(side_a.get("message"))), ["h3", "diff"], _pre_blob(_blob_text(side_a.get("diff"))), - ["h2", f"side B — {side_b.get('contributor', '?')}"], + ["h2", f"side B — {side_b.get('commit_id', side_b.get('contributor', '?'))}"], ["h3", "message"], _pre_blob(_blob_text(side_b.get("message"))), ["h3", "diff"], @@ -3156,9 +3222,9 @@ async def watch(): ], ["p.note", ( - "Pairwise council voting is ready." + "Pairwise council ranks every eligible commit." if key_ok else - "Single-contributor epochs can finalize, but contested rankings require OPENROUTER_API_KEY." + "Epochs with two or more eligible commits require OPENROUTER_API_KEY." ) ], ], diff --git a/tests/integration.clj b/tests/integration.clj index 11140e95a8cd86768a661a30e6b29a3ddfc95fcc..39b2476cb5ddf80733769f61343c81ba7287a974 100644 --- a/tests/integration.clj +++ b/tests/integration.clj @@ -507,7 +507,7 @@ (bind or-state @(:state or-mock)) (assert! (pos? (:model-requests or-state)) "OpenRouter /models was called") - (assert! (>= (:compare-requests or-state) 3) "at least 3 pairwise LLM calls (2 authors × 3 models)") + (assert! (>= (:compare-requests or-state) 3) "at least 3 pairwise LLM calls (2 commits × 3 models)") (bind ledger2 (get-json base-url "/api/ledger")) (assert! (>= (count ledger2) 3) @@ -543,6 +543,8 @@ "epoch page links comparisons") (assert! (str/includes? epoch-html "/judgments/") "epoch page links judgments") + (assert! (str/includes? epoch-html "commit ranking") + "epoch page shows per-commit ranking") (bind commit-href (second (re-find #"/commits/(c_[a-f0-9]+)" epoch-html))) (assert! (some? commit-href) "found a commit id on epoch page") diff --git a/tests/test_git_discovery.py b/tests/test_git_discovery.py index 00d26bfa85737b178c8822804e278211974e0efe..0f002a58bd9a122d41c5a85e32c16ed2b4404d2f 100644 --- a/tests/test_git_discovery.py +++ b/tests/test_git_discovery.py @@ -434,7 +434,7 @@ def test_emission_distribution_sums_exactly_to_total( assert entry.discovery_snapshot_id == "ranked-snapshot" -def test_single_contributor_ranking_is_total_and_uses_no_pairwise_votes( +def test_single_commit_ranking_skips_pairwise( discovery_config, monkeypatch, ): monkeypatch.setattr(c, "store", c.JsonlStore(discovery_config / "ledger.jsonl")) @@ -454,6 +454,65 @@ def test_single_contributor_ranking_is_total_and_uses_no_pairwise_votes( assert info["ranking_event_id"] +def test_same_contributor_multiple_commits_runs_pairwise( + discovery_config, monkeypatch, +): + monkeypatch.setattr(c, "store", c.JsonlStore(discovery_config / "ledger.jsonl")) + calls = {"n": 0} + + async def models(n=3): + return ["m1", "m2", "m3"] + + async def compare(model_id, side_a, side_b, **kwargs): + calls["n"] += 1 + assert "commit_id" in side_a and "commit_id" in side_b + if kwargs.get("persist"): + attempt_id = c.attempt_id_for(kwargs["comparison_id"], model_id, 1) + await c.append_evidence(kwargs["epoch"], "llm.judgment", { + "judgment_id": c.judgment_id_for({ + "attempt_id": attempt_id, + "comparison_id": kwargs["comparison_id"], + "model_id": model_id, + "winner": "A", + "ratio": "2:1", + "explanation": "ok", + }), + "attempt_id": attempt_id, + "comparison_id": kwargs["comparison_id"], + "model_id": model_id, + "winner": "A", + "ratio": "2:1", + "explanation": "ok", + "summary": "ok", + }) + return {"winner": "A", "ratio": "2:1", "explanation": "ok"} + + monkeypatch.setattr(c, "fetch_top_models", models) + monkeypatch.setattr(c, "llm_pairwise_compare", compare) + monkeypatch.setattr(c, "OPENROUTER_API_KEY", "test-key") + commits = [ + { + "contributor": "alice", + "oid": "sha1:" + char * 40, + "message": f"msg-{char}", + "patch": f"patch-{char}", + } + for char in ("a", "b", "c") + ] + ranking, used, info = asyncio.run(c.rank_commits(commits, epoch=0)) + assert set(ranking) == {"alice"} + assert ranking["alice"] > 0 + assert used == ["m1", "m2", "m3"] + assert calls["n"] >= 3 + completed = next( + e for e in c.store.read() + if isinstance(e, c.Evidence) and e.kind == "ranking.completed" + ) + assert len(completed.payload["commit_ranking"]) == 3 + assert "alice" in completed.payload["contributor_ranking"] + assert info["ranking_event_id"] + + def test_any_council_failure_aborts_ranking(discovery_config, monkeypatch): monkeypatch.setattr(c, "store", c.JsonlStore(discovery_config / "ledger.jsonl")) @@ -479,16 +538,16 @@ def test_any_council_failure_aborts_ranking(discovery_config, monkeypatch): asyncio.run(c.rank_commits(commits, epoch=0)) -def test_contested_ranking_requires_openrouter_key(monkeypatch): +def test_multi_commit_ranking_requires_openrouter_key(monkeypatch): monkeypatch.setattr(c, "OPENROUTER_API_KEY", "") commits = [ { - "contributor": contributor, + "contributor": "alice", "oid": "sha1:" + char * 40, - "message": contributor, + "message": char, "patch": "patch", } - for contributor, char in [("alice", "a"), ("bob", "b")] + for char in ("a", "b") ] with pytest.raises(RuntimeError, match="OPENROUTER_API_KEY"): asyncio.run(c.rank_commits(commits))