You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [5ca518f6] url refactor Side A — unified diff (full patch): diff --git a/Cargo.lock b/Cargo.lock index 67a09a3b54f778fa7e857fdd589c3ed9c92e1322..ad7e4fe6d4ba2f2b033916194c1ef1ed873f1d46 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1757,6 +1757,7 @@ name = "slug-types" version = "0.1.0" dependencies = [ "serde", + "url", ] [[package]] @@ -2272,6 +2273,7 @@ dependencies = [ "idna", "percent-encoding", "serde", + "serde_derive", ] [[package]] diff --git a/server/src/api/ui_html.rs b/server/src/api/ui_html.rs index 606f7d6f97a4428efb90d1e0d544934c861fc4c5..cd3e0f0afd972d9ad9e7e4b92c5fa4c22bb8f620 100644 --- a/server/src/api/ui_html.rs +++ b/server/src/api/ui_html.rs @@ -270,10 +270,10 @@ fn post_redirect_location(room: &str, thread_tag: &str) -> String { format!("/t/{tag}") } else { let room = room.trim(); - let Some((a, b)) = room.split_once('/') else { + let Some(seg) = slug_types::room_route_segment(room) else { return "/".to_string(); }; - format!("/r/{a}/{b}/t/{tag}") + format!("/r/{seg}/t/{tag}") } } diff --git a/server/src/api/write_actor.rs b/server/src/api/write_actor.rs index cb78d2f3f95b1bc163c1fb064d1d5f657e18000f..f9c3b8bd3fbf8fcb9c035e1a1572fef0b08fa8a9 100644 --- a/server/src/api/write_actor.rs +++ b/server/src/api/write_actor.rs @@ -19,13 +19,15 @@ use crate::{ use super::auth::{issue_token_for_user, verify_token}; use super::helpers::{now_ms, resolve_item}; use super::validate::{normalize_room_and_thread, validate_ingest_document}; -use slug_types::RpcResult; +use slug_types::{room_route_segment, RpcResult, ROOM_SHORT_ID_LEN}; fn gen_short_id() -> String { use rand::Rng; const ALPHABET: &[u8] = b"0123456789abcdefghijklmnopqrstuvwxyz"; let mut rng = rand::thread_rng(); - (0..7).map(|_| ALPHABET[rng.gen_range(0..ALPHABET.len())] as char).collect() + (0..ROOM_SHORT_ID_LEN) + .map(|_| ALPHABET[rng.gen_range(0..ALPHABET.len())] as char) + .collect() } fn parse_capability(s: &str) -> Result { @@ -55,8 +57,8 @@ async fn broadcast_web_refresh(state: &AppState, room_key: &str, thread_id: &str let feed_id = if room_key == "public" { "thread-feed" } else { "room-thread-feed" }; let thread_url = if room_key == "public" { format!("/t/{thread_id}") - } else if let Some((short, slug)) = room_key.split_once('/') { - format!("/r/{short}/{slug}/t/{thread_id}") + } else if let Some(seg) = room_route_segment(room_key) { + format!("/r/{seg}/t/{thread_id}") } else { format!("/t/{thread_id}") }; @@ -78,8 +80,8 @@ async fn broadcast_web_refresh(state: &AppState, room_key: &str, thread_id: &str let js = builder.build(); let mut path_prefixes = vec![if room_key == "public" { "/".to_string() - } else if let Some((short, slug)) = room_key.split_once('/') { - format!("/r/{short}/{slug}") + } else if let Some(seg) = room_route_segment(room_key) { + format!("/r/{seg}") } else { "/".to_string() }]; diff --git a/server/src/html/forum/nav.rs b/server/src/html/forum/nav.rs index 0ee33d91160fc5542817b5e3e9ab4fee1d0e600f..48fe11e46731670874ff8b6b05baa6f09ae0b7e4 100644 --- a/server/src/html/forum/nav.rs +++ b/server/src/html/forum/nav.rs @@ -1,7 +1,8 @@ use crate::canonical_path::canonicalize_item; use crate::reducer::ScopeId; +use slug_types::room_route_segment; -/// URL helpers for public `/t/…` and private room threads `/r/{short}/{slug}/t/…`. +/// URL helpers for public `/t/…` and private room threads `/r/{short}{slug}/t/…`. #[derive(Clone)] pub struct ThreadNav { pub room_wire: String, @@ -22,18 +23,15 @@ impl ThreadNav { } } - /// `room_id` wire form `shortid/slug`. + /// `room_id` wire form `shortid/slug` (HTTP uses [`slug_types::room_route_segment`]). pub(crate) fn from_room_id(room_id: &str) -> Option { - let (short, slug) = room_id.split_once('/')?; - if short.is_empty() || slug.is_empty() { - return None; - } + let room_seg = room_route_segment(room_id)?; Some(Self { room_wire: room_id.to_string(), scope: ScopeId::Room(room_id.to_string()), - room_path: format!("/r/{short}/{slug}"), - thread_path_prefix: format!("/r/{short}/{slug}/t"), - garden_path_prefix: format!("/r/{short}/{slug}/~"), + room_path: format!("/r/{room_seg}"), + thread_path_prefix: format!("/r/{room_seg}/t"), + garden_path_prefix: format!("/r/{room_seg}/~"), }) } diff --git a/server/src/html/forum/post_single.rs b/server/src/html/forum/post_single.rs index c316f8f836df9d4ef9c05ebd9e54f699540e6d72..473747b3da3d4d7a54b5e0c63165d2533df643e1 100644 --- a/server/src/html/forum/post_single.rs +++ b/server/src/html/forum/post_single.rs @@ -93,12 +93,14 @@ pub async fn thread_post_view( pub async fn room_thread_post_view( State(state): State, - Path((room_short, room_slug, tag, index_str)): Path<(String, String, String, String)>, + Path((room_key, tag, index_str)): Path<(String, String, String)>, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let reduced = state.reduced.read().await; let user = optional_principal(&headers, &jar, &reduced); if !user_can_view_room(&reduced, &room_id, user.as_deref()) { diff --git a/server/src/html/forum/views.rs b/server/src/html/forum/views.rs index be5df1745ef580891a167c23c3dd6c06f804f295..1ec421f84335cbfe7f9db775b73a8ed24b197174 100644 --- a/server/src/html/forum/views.rs +++ b/server/src/html/forum/views.rs @@ -183,16 +183,18 @@ pub async fn thread_view( thread_view_inner(state, tag, q, ThreadNav::public(), headers, jar, uri).await } -/// Room thread — `/r/:short/:slug/t/:tag` +/// Room thread — `/r/:room_key/t/:tag` (`room_key` = `{short}{slug}`). pub async fn room_thread_view( State(state): State, - Path((room_short, room_slug, tag)): Path<(String, String, String)>, + Path((room_key, tag)): Path<(String, String)>, Query(q): Query, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let reduced = state.reduced.read().await; let user = optional_principal(&headers, &jar, &reduced); if !user_can_view_room(&reduced, &room_id, user.as_deref()) { @@ -226,15 +228,17 @@ pub(super) fn room_not_found_page(jar: &CookieJar, uri: &Uri) -> impl IntoRespon (StatusCode::NOT_FOUND, Html(page.into_string())) } -/// Private room index — `/r/:short/:slug` +/// Private room index — `/r/:room_key` pub async fn room_page( State(state): State, - Path((room_short, room_slug)): Path<(String, String)>, + Path(room_key): Path, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "room not found").into_response(); + }; let now = now_ms(); let reduced = state.reduced.read().await; if !reduced.rooms.contains(&room_id) { @@ -266,7 +270,10 @@ pub async fn room_page( let audit_cli = format!("npx slugsocial private {room_id} audit"); drop(reduced); - let slug_display = room_slug.as_str(); + let slug_display = room_id + .split_once('/') + .map(|(_, slug)| slug) + .unwrap_or(room_id.as_str()); let page = layout( &format!("room {slug_display} — slug.social"), "view-thread", diff --git a/server/src/html/garden.rs b/server/src/html/garden.rs index 423f23fd8c9ad7b7f454d6ea7a9a7607a4c9c5b9..e615dd356bcf634232d85610c0a26235ead125fd 100644 --- a/server/src/html/garden.rs +++ b/server/src/html/garden.rs @@ -309,12 +309,14 @@ pub async fn external_ontology_path( pub async fn room_garden_index( State(state): State, - Path((room_short, room_slug)): Path<(String, String)>, + Path(room_key): Path, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let Some(nav) = ThreadNav::from_room_id(&room_id) else { return (StatusCode::NOT_FOUND, "bad room path").into_response(); }; @@ -341,12 +343,14 @@ pub async fn room_garden_index( pub async fn room_external_garden_index( State(state): State, - Path((room_short, room_slug)): Path<(String, String)>, + Path(room_key): Path, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let Some(nav) = ThreadNav::from_room_id(&room_id) else { return (StatusCode::NOT_FOUND, "bad room path").into_response(); }; @@ -416,12 +420,14 @@ pub async fn room_external_garden_index( pub async fn room_external_ontology_path( State(state): State, - Path((room_short, room_slug, path)): Path<(String, String, String)>, + Path((room_key, path)): Path<(String, String)>, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let Some(nav) = ThreadNav::from_room_id(&room_id) else { return (StatusCode::NOT_FOUND, "bad room path").into_response(); }; @@ -442,12 +448,14 @@ pub async fn room_external_ontology_path( pub async fn room_ontology_path( State(state): State, - Path((room_short, room_slug, path)): Path<(String, String, String)>, + Path((room_key, path)): Path<(String, String)>, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let Some(nav) = ThreadNav::from_room_id(&room_id) else { return (StatusCode::NOT_FOUND, "bad room path").into_response(); }; diff --git a/server/src/html/search.rs b/server/src/html/search.rs index 43e6ebf36cf0fe72c96f0f9d850bea51ac094c43..f01732f7edb32c68fcc10c39545c8f56476adb27 100644 --- a/server/src/html/search.rs +++ b/server/src/html/search.rs @@ -351,8 +351,8 @@ fn render_search_results(results: &SearchResults, query: &str) -> Markup { ul class="search-posts" { @for r in &results.posts { @let (post_href, post_label) = if let Some((room, tag)) = r.thread.split_once("/#") { - if let Some((short, slug)) = room.split_once('/') { - (format!("/r/{short}/{slug}/t/{tag}"), format!("{room}/#{tag}")) + if let Some(seg) = slug_types::room_route_segment(room) { + (format!("/r/{seg}/t/{tag}"), format!("{room}/#{tag}")) } else { ("/".to_string(), r.thread.clone()) } diff --git a/server/src/lib.rs b/server/src/lib.rs index 9e69c3f4a153527caa77b00ad101aca94b425995..71032f9a0b556097d90f3b403c677e0c86429ee4 100644 --- a/server/src/lib.rs +++ b/server/src/lib.rs @@ -79,30 +79,30 @@ pub fn create_app(state: AppState) -> Router { .route("/~/*path", get(crate::html::ontology_path)) .route("/-", get(crate::html::external_garden_index)) .route("/-/*path", get(crate::html::external_ontology_path)) - .route("/r/:room_short/:room_slug/~", get(crate::html::room_garden_index)) + .route("/r/:room_key/~", get(crate::html::room_garden_index)) .route( - "/r/:room_short/:room_slug/~/*path", + "/r/:room_key/~/*path", get(crate::html::room_ontology_path), ) .route( - "/r/:room_short/:room_slug/-", + "/r/:room_key/-", get(crate::html::room_external_garden_index), ) .route( - "/r/:room_short/:room_slug/-/*path", + "/r/:room_key/-/*path", get(crate::html::room_external_ontology_path), ) .route("/t/:tag/:index", get(crate::html::thread_post_view)) .route("/t/:tag", get(crate::html::thread_view)) .route( - "/r/:room_short/:room_slug/t/:thread_tag/:index", + "/r/:room_key/t/:thread_tag/:index", get(crate::html::room_thread_post_view), ) .route( - "/r/:room_short/:room_slug/t/:thread_tag", + "/r/:room_key/t/:thread_tag", get(crate::html::room_thread_view), ) - .route("/r/:room_short/:room_slug", get(crate::html::room_page)) + .route("/r/:room_key", get(crate::html::room_page)) .route("/join/:token", get(api::get_join_invite)) .route("/auth/login", get(api::get_auth_login)) .route("/auth/callback", get(api::get_auth_callback)) diff --git a/server/tests/integration.rs b/server/tests/integration.rs index 515c691d04ef3f1f4bdfa8ee911360a43fd2b26e..feb73d475b3fc635d7f16c329d4600718c7472ba 100644 --- a/server/tests/integration.rs +++ b/server/tests/integration.rs @@ -1,4 +1,5 @@ use sha2::{Digest, Sha256}; +use slug_types::room_route_segment; use slugsocial_server::{ event_log::EventLog, events::{Event, TokenIssued, UserRegistered}, @@ -607,7 +608,7 @@ async fn test_private_room_thread_urls_use_t_segment() { .as_str() .unwrap() .to_string(); - let (room_short, room_slug) = room_id.split_once('/').unwrap(); + let room_seg = room_route_segment(&room_id).unwrap(); let rpc = ui_post_ingest_rpc(&room_id, "main-thread", "private post via web"); let post = client @@ -626,7 +627,7 @@ async fn test_private_room_thread_urls_use_t_segment() { Some("text/javascript; charset=utf-8") ); let post_js = post.text().await.unwrap(); - let location = format!("/r/{room_short}/{room_slug}/t/main-thread"); + let location = format!("/r/{room_seg}/t/main-thread"); assert!(post_js.contains(&format!("window.location = {:?};", location))); assert!(post_js.contains("#room-thread-feed")); assert!(post_js.contains("#thread-feed-region")); @@ -665,7 +666,7 @@ async fn test_private_room_post_links_use_private_garden_routes() { .as_str() .unwrap() .to_string(); - let (room_short, room_slug) = room_id.split_once('/').unwrap(); + let room_seg = room_route_segment(&room_id).unwrap(); let rpc = ui_post_ingest_rpc( &room_id, @@ -682,18 +683,18 @@ async fn test_private_room_post_links_use_private_garden_routes() { assert_eq!(post.status(), reqwest::StatusCode::OK); let thread_page = client - .get(format!("http://{addr}/r/{room_short}/{room_slug}/t/garden-thread")) + .get(format!("http://{addr}/r/{room_seg}/t/garden-thread")) .header("Authorization", format!("Bearer {bearer}")) .send() .await .unwrap(); assert!(thread_page.status().is_success()); let body = thread_page.text().await.unwrap(); - assert!(body.contains(&format!("/r/{room_short}/{room_slug}/~/secret/item"))); + assert!(body.contains(&format!("/r/{room_seg}/~/secret/item"))); assert!(!body.contains("href=\"/~/secret/item\"")); let garden_page = client - .get(format!("http://{addr}/r/{room_short}/{room_slug}/~/secret/item")) + .get(format!("http://{addr}/r/{room_seg}/~/secret/item")) .header("Authorization", format!("Bearer {bearer}")) .send() .await @@ -701,7 +702,7 @@ async fn test_private_room_post_links_use_private_garden_routes() { assert!(garden_page.status().is_success()); let garden_body = garden_page.text().await.unwrap(); assert!(garden_body.contains("classified")); - assert!(garden_body.contains(&format!("/r/{room_short}/{room_slug}/t/garden-thread"))); + assert!(garden_body.contains(&format!("/r/{room_seg}/t/garden-thread"))); } #[tokio::test] @@ -726,7 +727,7 @@ async fn test_private_room_garden_root_lists_top_level_tilde_children() { .as_str() .unwrap() .to_string(); - let (room_short, room_slug) = room_id.split_once('/').unwrap(); + let room_seg = room_route_segment(&room_id).unwrap(); let rpc = ui_post_ingest_rpc( &room_id, @@ -743,7 +744,7 @@ async fn test_private_room_garden_root_lists_top_level_tilde_children() { assert_eq!(post.status(), reqwest::StatusCode::OK); let root_page = client - .get(format!("http://{addr}/r/{room_short}/{room_slug}/~")) + .get(format!("http://{addr}/r/{room_seg}/~")) .header("Authorization", format!("Bearer {bearer}")) .send() .await @@ -782,10 +783,10 @@ async fn test_empty_private_room_garden_returns_404() { .as_str() .unwrap() .to_string(); - let (room_short, room_slug) = room_id.split_once('/').unwrap(); + let room_seg = room_route_segment(&room_id).unwrap(); let root = client - .get(format!("http://{addr}/r/{room_short}/{room_slug}/~")) + .get(format!("http://{addr}/r/{room_seg}/~")) .header("Authorization", format!("Bearer {bearer}")) .send() .await @@ -896,8 +897,8 @@ async fn test_sse_stream_emits_evalable_js_after_post() { .as_str() .unwrap() .to_string(); - let (room_short, room_slug) = room_id.split_once('/').unwrap(); - let room_path = format!("/r/{room_short}/{room_slug}"); + let room_seg = room_route_segment(&room_id).unwrap(); + let room_path = format!("/r/{room_seg}"); let sse_resp = client .get(format!("http://{addr}/sse?path={}", urlencoding::encode(&room_path))) diff --git a/test/browser_room_delete.clj b/test/browser_room_delete.clj index 833e3f8aa7f850b60ee59742736143cbf101a07d..764ecaa674798a61c1ffca3859f2a56a91a0e97e 100644 --- a/test/browser_room_delete.clj +++ b/test/browser_room_delete.clj @@ -54,7 +54,7 @@ room-id (get-in create-json ["results" 0 "result" "RoomCreated" "room_id"]) _ (is (string? room-id) "room id present") [room-short room-slug] (str/split room-id #"/" 2) - room-path (str "/r/" room-short "/" room-slug)] + room-path (str "/r/" room-short room-slug)] (core/with-playwright [pw] (core/with-browser [browser (core/launch-chromium pw {:headless true :channel "chrome"})] (core/with-context [ctx (core/new-context browser)] diff --git a/test/browser_sse.clj b/test/browser_sse.clj index 8bb064dbc2dc0feb6caaa67f520a461645958277..678399b79f78ff48b5e77342fa21e69992393834 100644 --- a/test/browser_sse.clj +++ b/test/browser_sse.clj @@ -77,7 +77,7 @@ (login-user! bob-pg base-url "bob") (let [[room-short room-slug] (str/split room-id #"/" 2) - room-url (str base-url "/r/" room-short "/" room-slug) + room-url (str base-url "/r/" room-short room-slug) thread-url (str room-url "/t/sse-thread")] ;; Object under test: slug_ui.js intercepts POST /ui, evals JS, morphs ;; #new-thread-ui-slot (expand compose), then post_ingest redirects to thread. diff --git a/test/walkthrough_fixture.clj b/test/walkthrough_fixture.clj index a487ff1ef8548580577914fb33f321586859c9af..f7eb1cacca0317129e9baf68fa4f4e33356ad70f 100644 --- a/test/walkthrough_fixture.clj +++ b/test/walkthrough_fixture.clj @@ -88,9 +88,9 @@ :room {:id room-id :short room-short :slug room-slug - :url (str base-url "/r/" room-short "/" room-slug) - :thread_url (str base-url "/r/" room-short "/" room-slug "/t/walkthrough-thread") - :garden_url (str base-url "/r/" room-short "/" room-slug "/~/secret/item")}})) + :url (str base-url "/r/" room-short room-slug) + :thread_url (str base-url "/r/" room-short room-slug "/t/walkthrough-thread") + :garden_url (str base-url "/r/" room-short room-slug "/~/secret/item")}})) (defn- rebase-fixture-summary [saved current-base-url current-google-url current-data-dir] (let [inner (:summary saved) @@ -103,9 +103,9 @@ :data_dir (str current-data-dir) :summary (assoc inner :room (assoc room - :url (str current-base-url "/r/" rs "/" lg) - :thread_url (str current-base-url "/r/" rs "/" lg "/t/walkthrough-thread") - :garden_url (str current-base-url "/r/" rs "/" lg "/~/secret/item")))))) + :url (str current-base-url "/r/" rs lg) + :thread_url (str current-base-url "/r/" rs lg "/t/walkthrough-thread") + :garden_url (str current-base-url "/r/" rs lg "/~/secret/item")))))) (defn- fixture-log-present? [data-dir] (let [p (fs/path data-dir "events.jsonl")] diff --git a/types/Cargo.toml b/types/Cargo.toml index 1ed06d483ac7c126516bce27f90e7639c2dfe370..5dc3becf238f23243313dff0484c682a7b225737 100644 --- a/types/Cargo.toml +++ b/types/Cargo.toml @@ -5,3 +5,4 @@ edition = "2021" [dependencies] serde = { version = "1.0", features = ["derive"] } +url = { version = "2.5", features = ["serde"] } diff --git a/types/src/lib.rs b/types/src/lib.rs index fce1b5fa5e259a4b86a11cb4d4f7cc9bec3a0aa1..516cf935f15fca97081b39b988da5be894c67725 100644 --- a/types/src/lib.rs +++ b/types/src/lib.rs @@ -1,5 +1,7 @@ use serde::{Deserialize, Serialize}; +pub mod room_route; +pub mod url_normalize; pub mod paths; pub mod timeago; @@ -8,6 +10,8 @@ pub use paths::{ CanonicalItemUrl, ForumThreadUrl, GardenItemUrl, RelativePath, SLUG_TILDE_ONTOLOGY_ROOT, TildeHttpPathTail, TildeOntologyPath, TildePath, tilde_http_path_to_canonical, }; +pub use room_route::{room_id_from_route_segment, room_route_segment, ROOM_SHORT_ID_LEN}; +pub use url_normalize::normalize_http_identity_url; /// Max characters returned for a garden item body unless `full=true` / `--full` (API + CLI). pub const MAX_ITEM_BODY_PREVIEW_CHARS: usize = 100_000; @@ -657,3 +661,6 @@ pub struct VoteResponse { pub ranking: Vec, pub next: NextMoves, } + +#[cfg(test)] +mod url_identity_tests; diff --git a/types/src/paths.rs b/types/src/paths.rs index 8971f599a7ebcf399f60c306d35ee28d0f581194..98787a5fb481a1599556564bbbbd1d54dc693fc3 100644 --- a/types/src/paths.rs +++ b/types/src/paths.rs @@ -5,7 +5,7 @@ //! //! - **[`canonicalize_item`] / [`CanonicalItemUrl`]** — graph storage key and DSL form; tilde //! ontology root is always [`SLUG_TILDE_ONTOLOGY_ROOT`] (no `…/~/` trailing slash only). -//! - **[`TildeHttpPathTail`]** — capture from `GET /~/*path` or `…/r/…/~/…` (the `*path` segment). +//! - **[`TildeHttpPathTail`]** — capture from `GET /~/*path` or `…/r/{short}{slug}/~/…` (the `*path` segment). //! - **`-/…` wire form** — external items; see [`canonicalize_item`] dash branch. //! - **[`GardenItemUrl`], [`ForumThreadUrl`]** — JSON / browser href surfaces. @@ -15,6 +15,9 @@ use std::ops::Deref; use serde::{Deserialize, Serialize}; +use crate::room_route::room_route_segment; +use crate::url_normalize::{host_preserves_dash_path_case, normalize_http_identity_url}; + // --------------------------------------------------------------------------- // Slug tilde ontology (single storage form for `~/`) // --------------------------------------------------------------------------- @@ -41,6 +44,29 @@ pub fn canonicalize_tag(input: &str) -> String { input.trim().trim_start_matches('#').to_lowercase() } +fn finalize_external_identity_url(s: String) -> String { + if s.starts_with("https://slug.social/") { + return s; + } + let normalized = normalize_http_identity_url(&s).unwrap_or_else(|| s.clone()); + strip_redundant_root_slash(&normalized).unwrap_or(normalized) +} + +/// `url::Url` serializes bare hosts with a `/` path; we keep host-only items slash-free for stable +/// keys matching the pre-normalizer spellings. +fn strip_redundant_root_slash(s: &str) -> Option { + let u = url::Url::parse(s).ok()?; + if u.path() == "/" && u.query().is_none() && u.fragment().is_none() { + let scheme = u.scheme(); + let host = u.host_str()?; + return Some(match u.port() { + Some(p) => format!("{scheme}://{host}:{p}"), + None => format!("{scheme}://{host}"), + }); + } + None +} + /// Ontology item reference → canonical absolute URL on the slug host. pub fn canonicalize_item(input: &str) -> String { let s = input.trim(); @@ -57,8 +83,9 @@ pub fn canonicalize_item(input: &str) -> String { if host.is_empty() { return String::new(); } + let preserve_case = host_preserves_dash_path_case(&host); return if tail.is_empty() { - format!("https://{}", host) + finalize_external_identity_url(format!("https://{}", host)) } else { let path = tail .trim_start_matches('/') @@ -68,33 +95,35 @@ pub fn canonicalize_item(input: &str) -> String { let t = seg.trim(); if t.is_empty() { None + } else if preserve_case { + Some(t.to_string()) } else { Some(t.to_lowercase()) } }) .collect::>() .join("/"); - format!("https://{}/{}", host, path) + finalize_external_identity_url(format!("https://{}/{}", host, path)) }; } if let Some(rest) = s.strip_prefix("https://") { let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); let host = host.trim().to_lowercase(); - if tail.is_empty() { - return format!("https://{}", host); + return finalize_external_identity_url(if tail.is_empty() { + format!("https://{}", host) } else { - return format!("https://{}/{}", host, tail); - } + format!("https://{}/{}", host, tail) + }); } if let Some(rest) = s.strip_prefix("http://") { let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); let host = host.trim().to_lowercase(); - if tail.is_empty() { - return format!("http://{}", host); + return finalize_external_identity_url(if tail.is_empty() { + format!("http://{}", host) } else { - return format!("http://{}/{}", host, tail); - } + format!("http://{}/{}", host, tail) + }); } let is_tilde = s.starts_with("~/"); @@ -319,7 +348,7 @@ impl CanonicalItemUrl { } } -/// HTTP route capture: path segment after `~/` in `GET /~/*path` or `…/r/…/~/…` (empty = ontology root). +/// HTTP route capture: path segment after `~/` in `GET /~/*path` or `…/r/{short}{slug}/~/…` (empty = ontology root). #[derive(Debug, Clone, PartialEq, Eq, Hash)] pub struct TildeHttpPathTail(pub String); @@ -506,12 +535,9 @@ fn garden_href_string(item: &str, room_wire: &str) -> String { if room.is_empty() || room == "public" { return api_path_or_url(item); } - let Some((short, slug)) = room.split_once('/') else { + let Some(room_seg) = room_route_segment(room) else { return api_path_or_url(item); }; - if short.is_empty() || slug.is_empty() { - return api_path_or_url(item); - } let Some(c) = CanonicalItemUrl::parse(item) else { return api_path_or_url(item); }; @@ -520,19 +546,19 @@ fn garden_href_string(item: &str, room_wire: &str) -> String { let root_norm = root.as_str().trim_end_matches('/'); if let Some(tail) = c.tilde_tail() { return if tail.is_empty() { - format!("https://slug.social/r/{short}/{slug}/~") + format!("https://slug.social/r/{room_seg}/~") } else { - format!("https://slug.social/r/{short}/{slug}/~/{}", tail) + format!("https://slug.social/r/{room_seg}/~/{}", tail) }; } if item_norm == root_norm { - return format!("https://slug.social/r/{short}/{slug}/~"); + return format!("https://slug.social/r/{room_seg}/~"); } // External http(s) items use the same `/-/…` namespace under the room garden. if c.as_str().starts_with("https://") || c.as_str().starts_with("http://") { let tail = c.display_path(); let tail = tail.strip_prefix("-/").unwrap_or(tail.as_str()); - return format!("https://slug.social/r/{short}/{slug}/-/{tail}"); + return format!("https://slug.social/r/{room_seg}/-/{tail}"); } api_path_or_url(item) } @@ -556,12 +582,8 @@ impl ForumThreadUrl { let tag = thread_tag.trim().trim_start_matches('#'); Self(if room.is_empty() || room == "public" { format!("https://slug.social/t/{tag}") - } else if let Some((short, slug)) = room.split_once('/') { - if short.is_empty() || slug.is_empty() { - format!("https://slug.social/t/{tag}") - } else { - format!("https://slug.social/r/{short}/{slug}/t/{tag}") - } + } else if let Some(room_seg) = room_route_segment(room) { + format!("https://slug.social/r/{room_seg}/t/{tag}") } else { format!("https://slug.social/t/{tag}") }) @@ -706,7 +728,7 @@ mod tests { fn garden_private_room_prefixes_ontology() { assert_eq!( GardenItemUrl::from_storage_str("https://slug.social/~/topic/x", "9ab12cd/my-room").as_str(), - "https://slug.social/r/9ab12cd/my-room/~/topic/x" + "https://slug.social/r/9ab12cdmy-room/~/topic/x" ); } @@ -714,11 +736,11 @@ mod tests { fn garden_private_room_ontology_root() { assert_eq!( GardenItemUrl::from_storage_str("https://slug.social/~", "9ab12cd/my-room").as_str(), - "https://slug.social/r/9ab12cd/my-room/~" + "https://slug.social/r/9ab12cdmy-room/~" ); assert_eq!( GardenItemUrl::from_storage_str("https://slug.social/~/", "9ab12cd/my-room").as_str(), - "https://slug.social/r/9ab12cd/my-room/~" + "https://slug.social/r/9ab12cdmy-room/~" ); } @@ -727,7 +749,7 @@ mod tests { let u = "https://example.com/z"; assert_eq!( GardenItemUrl::from_storage_str(u, "9ab12cd/my-room").as_str(), - "https://slug.social/r/9ab12cd/my-room/-/example.com/z" + "https://slug.social/r/9ab12cdmy-room/-/example.com/z" ); } @@ -739,7 +761,19 @@ mod tests { ); assert_eq!( ForumThreadUrl::from_room_tag("9ab12cd/my-room", "#debate").as_str(), - "https://slug.social/r/9ab12cd/my-room/t/debate" + "https://slug.social/r/9ab12cdmy-room/t/debate" + ); + } + + #[test] + fn canonicalize_youtube_short_links() { + assert_eq!( + canonicalize_item("https://youtu.be/dQw4w9WgXcQ"), + "https://www.youtube.com/watch?v=dQw4w9WgXcQ" + ); + assert_eq!( + canonicalize_item("-/youtu.be/dQw4w9WgXcQ"), + "https://www.youtube.com/watch?v=dQw4w9WgXcQ" ); } diff --git a/types/src/room_route.rs b/types/src/room_route.rs new file mode 100644 index 0000000000000000000000000000000000000000..4f4780c88e30f2b28e2cfcd7aee713d39dbd1c66 --- /dev/null +++ b/types/src/room_route.rs @@ -0,0 +1,56 @@ +//! HTTP path encoding for private rooms: `/r/{short}{slug}` (short is fixed width). + +/// Byte length of the random `short` segment in `short/slug` room ids. +/// Must match room creation (`gen_short_id`) and [`super::paths`][] URL builders. +pub const ROOM_SHORT_ID_LEN: usize = 7; + +/// `ab12cde/my-room` → `ab12cdemy-room` for a single `/r/…` path segment. +pub fn room_route_segment(room_id: &str) -> Option { + let (short, slug) = room_id.split_once('/')?; + if short.len() != ROOM_SHORT_ID_LEN || short.is_empty() || slug.is_empty() { + return None; + } + if !short + .bytes() + .all(|b| matches!(b, b'0'..=b'9' | b'a'..=b'z')) + { + return None; + } + Some(format!("{short}{slug}")) +} + +/// `/r/{short}{slug}` path segment → `short/slug` wire id (inverse of [`room_route_segment`]). +pub fn room_id_from_route_segment(seg: &str) -> Option { + if seg.len() <= ROOM_SHORT_ID_LEN { + return None; + } + let (short, slug) = seg.split_at(ROOM_SHORT_ID_LEN); + if short.is_empty() || slug.is_empty() { + return None; + } + if !short + .bytes() + .all(|b| matches!(b, b'0'..=b'9' | b'a'..=b'z')) + { + return None; + } + Some(format!("{short}/{slug}")) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn round_trip_room_segment() { + let id = "9ab12cd/my-room"; + let seg = room_route_segment(id).unwrap(); + assert_eq!(seg, "9ab12cdmy-room"); + assert_eq!(room_id_from_route_segment(&seg).as_deref(), Some(id)); + } + + #[test] + fn too_short_segment_rejected() { + assert!(room_id_from_route_segment("9ab12cd").is_none()); + } +} diff --git a/types/src/url_identity_tests.rs b/types/src/url_identity_tests.rs new file mode 100644 index 0000000000000000000000000000000000000000..e2ff0a9a64a88049e31f20979c703777df6a9b65 --- /dev/null +++ b/types/src/url_identity_tests.rs @@ -0,0 +1,126 @@ +//! How `url::Url` behaves as `HashMap` keys (`Eq` + `Hash`). +//! +//! If `ItemId::External` stores `Url`, these tests are the contract you are buying into +//! (or the baseline before you add a custom normalization layer). + +use std::collections::HashMap; +use std::hash::{Hash, Hasher}; +use url::Url; + +fn hash_one(url: &Url) -> u64 { + let mut h = std::collections::hash_map::DefaultHasher::new(); + url.hash(&mut h); + h.finish() +} + +#[test] +fn identical_parse_strings_are_eq_and_share_hash_bucket() { + let a = Url::parse("https://example.com/path").unwrap(); + let b = Url::parse("https://example.com/path").unwrap(); + assert_eq!(a, b); + assert_eq!(hash_one(&a), hash_one(&b)); + + let mut m: HashMap = HashMap::new(); + m.insert(a, 1); + *m.entry(b).or_default() += 10; + assert_eq!(m.len(), 1); + assert_eq!(m[&Url::parse("https://example.com/path").unwrap()], 11); +} + +#[test] +fn host_is_ascii_lowercase_in_eq() { + let lower = Url::parse("https://examplE.com/").unwrap(); + let upper = Url::parse("https://EXAMPLE.com/").unwrap(); + assert_eq!(lower, upper); + assert_eq!(hash_one(&lower), hash_one(&upper)); +} + +#[test] +fn path_space_normalizes_to_percent_encoding_so_forms_merge() { + let encoded = Url::parse("https://example.com/a%20b").unwrap(); + let decoded = Url::parse("https://example.com/a b").unwrap(); + // Parser normalizes both to the same internal path (`/a%20b`). + assert_eq!(encoded, decoded); + assert_eq!(hash_one(&encoded), hash_one(&decoded)); + + let mut m: HashMap = HashMap::new(); + m.insert(encoded, "first"); + assert_eq!(m.insert(decoded, "second"), Some("first")); + assert_eq!(m.len(), 1); + assert_eq!(m.values().next().copied(), Some("second")); +} + +#[test] +fn encoded_slash_in_segment_stays_distinct_from_real_path_separator() { + let encoded = Url::parse("https://example.com/a%2Fb").unwrap(); + let real_slash = Url::parse("https://example.com/a/b").unwrap(); + assert_ne!(encoded, real_slash); + assert_ne!(hash_one(&encoded), hash_one(&real_slash)); +} + +#[test] +fn trailing_slash_on_path_is_significant_for_eq() { + let with_slash = Url::parse("https://example.com/foo/").unwrap(); + let no_slash = Url::parse("https://example.com/foo").unwrap(); + assert_ne!(with_slash, no_slash); + assert_ne!(hash_one(&with_slash), hash_one(&no_slash)); +} + +#[test] +fn default_http_port_80_is_normalized_in_representation() { + let explicit = Url::parse("http://example.com:80/").unwrap(); + let implicit = Url::parse("http://example.com/").unwrap(); + assert_eq!(explicit, implicit); + assert_eq!(hash_one(&explicit), hash_one(&implicit)); +} + +#[test] +fn default_https_port_443_is_normalized() { + let explicit = Url::parse("https://example.com:443/foo").unwrap(); + let implicit = Url::parse("https://example.com/foo").unwrap(); + assert_eq!(explicit, implicit); +} + +#[test] +fn non_default_port_is_part_of_identity() { + let a = Url::parse("https://example.com:444/").unwrap(); + let b = Url::parse("https://example.com:445/").unwrap(); + assert_ne!(a, b); +} + +#[test] +fn empty_path_vs_slash_only_path_may_differ() { + let root = Url::parse("https://example.com").unwrap(); + let slash = Url::parse("https://example.com/").unwrap(); + // Both serialize to `https://example.com/` in practice for this crate — verify. + assert_eq!(root, slash, "document: root and trailing-slash-only merge for this parser"); +} + +#[test] +fn scheme_case_is_normalized_to_lowercase() { + let lower = Url::parse("https://example.com/").unwrap(); + let upper = Url::parse("HTTPS://example.com/").unwrap(); + assert_eq!(lower, upper); +} + +#[test] +fn fragment_is_part_of_eq_and_hash() { + let no_frag = Url::parse("https://example.com/a").unwrap(); + let frag = Url::parse("https://example.com/a#section").unwrap(); + assert_ne!( + no_frag, frag, + "#fragment is included in PartialEq — anchors are different HashMap keys" + ); + assert_ne!(hash_one(&no_frag), hash_one(&frag)); +} + +#[test] +fn query_order_and_encoding_can_split_identity() { + let a = Url::parse("https://example.com/?b=2&a=1").unwrap(); + let b = Url::parse("https://example.com/?a=1&b=2").unwrap(); + assert_ne!(a, b, "query pairs order is preserved in serialization"); + + let plus = Url::parse("https://example.com/?q=a+b").unwrap(); + let encoded = Url::parse("https://example.com/?q=a%20b").unwrap(); + assert_ne!(plus, encoded, "space as + vs %20 — different keys unless normalized"); +} diff --git a/types/src/url_normalize.rs b/types/src/url_normalize.rs new file mode 100644 index 0000000000000000000000000000000000000000..d5bc913f33ad29cd0cd1c7158e28e4fefdbf5f5d --- /dev/null +++ b/types/src/url_normalize.rs @@ -0,0 +1,234 @@ +//! Normalization for external `http(s)://` item identity (not slug tilde ontology). +//! +//! Policy (intentional, extend here as new domains need treatment): +//! - Query pairs sorted lexicographically by **lowercased** key, then value. +//! - YouTube family → stable `www.youtube.com` shapes where possible. + +use url::Url; + +/// Whether `-/` → `https://…` path segments should keep original casing (YouTube video IDs are +/// case-sensitive). +pub(crate) fn host_preserves_dash_path_case(host: &str) -> bool { + let h = host.trim().to_ascii_lowercase(); + let h = h.strip_prefix("www.").unwrap_or(h.as_str()); + matches!( + h, + "youtu.be" | "youtube.com" | "m.youtube.com" | "music.youtube.com" + ) +} + +/// Normalize external http(s) URLs for stable [`super::paths::canonicalize_item`] output. +pub fn normalize_http_identity_url(s: &str) -> Option { + let mut u = Url::parse(s).ok()?; + if !matches!(u.scheme(), "http" | "https") { + return None; + } + rewrite_youtube(&mut u); + sort_query_pairs(&mut u); + Some(u.to_string()) +} + +fn base_host(host: &str) -> String { + let lower = host.to_ascii_lowercase(); + lower + .strip_prefix("www.") + .unwrap_or(lower.as_str()) + .to_string() +} + +fn rewrite_youtube(u: &mut Url) { + let Some(host_raw) = u.host_str() else { + return; + }; + let base = base_host(host_raw); + let path = u.path().to_string(); + + match base.as_str() { + "youtu.be" => { + let id = path.trim_start_matches('/').split('/').next().unwrap_or("").to_string(); + if id.is_empty() { + return; + } + let saved: Vec<(String, String)> = u.query_pairs().into_owned().collect(); + let Ok(mut out) = Url::parse(&format!("https://www.youtube.com/watch?v={id}")) else { + return; + }; + { + let mut q = out.query_pairs_mut(); + for (k, v) in saved { + if k.eq_ignore_ascii_case("v") { + continue; + } + q.append_pair(&k, &v); + } + } + *u = out; + } + "youtube.com" | "m.youtube.com" => { + if base == "m.youtube.com" { + let _ = u.set_host(Some("www.youtube.com")); + } + if path.starts_with("/embed/") { + let id = path + .strip_prefix("/embed/") + .unwrap_or("") + .trim_matches('/') + .split('/') + .next() + .unwrap_or("") + .to_string(); + if id.is_empty() { + return; + } + let saved: Vec<(String, String)> = u.query_pairs().into_owned().collect(); + let Ok(mut out) = Url::parse(&format!("https://www.youtube.com/watch?v={id}")) + else { + return; + }; + { + let mut q = out.query_pairs_mut(); + for (k, v) in saved { + if k.eq_ignore_ascii_case("v") { + continue; + } + q.append_pair(&k, &v); + } + } + *u = out; + return; + } + if path.starts_with("/v/") { + let id = path + .strip_prefix("/v/") + .unwrap_or("") + .trim_matches('/') + .split('/') + .next() + .unwrap_or("") + .to_string(); + if id.is_empty() { + return; + } + let saved: Vec<(String, String)> = u.query_pairs().into_owned().collect(); + let Ok(mut out) = Url::parse(&format!("https://www.youtube.com/watch?v={id}")) + else { + return; + }; + { + let mut q = out.query_pairs_mut(); + for (k, v) in saved { + if k.eq_ignore_ascii_case("v") { + continue; + } + q.append_pair(&k, &v); + } + } + *u = out; + return; + } + if path.starts_with("/watch") { + let _ = u.set_host(Some("www.youtube.com")); + return; + } + if path.starts_with("/shorts/") { + let id = path + .strip_prefix("/shorts/") + .unwrap_or("") + .trim_matches('/') + .split('/') + .next() + .unwrap_or("") + .to_string(); + if id.is_empty() { + return; + } + let saved: Vec<(String, String)> = u.query_pairs().into_owned().collect(); + let Ok(mut out) = Url::parse(&format!("https://www.youtube.com/shorts/{id}")) + else { + return; + }; + { + let mut q = out.query_pairs_mut(); + for (k, v) in saved { + q.append_pair(&k, &v); + } + } + *u = out; + return; + } + let _ = u.set_host(Some("www.youtube.com")); + } + "music.youtube.com" => {} + _ => {} + } +} + +fn sort_query_pairs(u: &mut Url) { + let pairs: Vec<(String, String)> = u.query_pairs().into_owned().collect(); + if pairs.is_empty() { + u.set_query(None); + return; + } + let mut pairs = pairs; + pairs.sort_by(|a, b| { + a.0.to_ascii_lowercase() + .cmp(&b.0.to_ascii_lowercase()) + .then_with(|| a.1.cmp(&b.1)) + }); + u.set_query(None); + { + let mut q = u.query_pairs_mut(); + for (k, v) in pairs { + q.append_pair(&k, &v); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn youtube_youtu_be_to_watch() { + assert_eq!( + normalize_http_identity_url("https://youtu.be/dQw4w9WgXcQ").as_deref(), + Some("https://www.youtube.com/watch?v=dQw4w9WgXcQ") + ); + } + + #[test] + fn youtube_watch_query_sorted() { + assert_eq!( + normalize_http_identity_url("https://youtube.com/watch?v=Z&a=1&b=2").as_deref(), + Some("https://www.youtube.com/watch?a=1&b=2&v=Z") + ); + assert_eq!( + normalize_http_identity_url("https://youtube.com/watch?b=2&a=1&v=Z").as_deref(), + Some("https://www.youtube.com/watch?a=1&b=2&v=Z") + ); + } + + #[test] + fn youtube_embed_to_watch() { + assert_eq!( + normalize_http_identity_url("https://www.youtube.com/embed/dQw4w9WgXcQ").as_deref(), + Some("https://www.youtube.com/watch?v=dQw4w9WgXcQ") + ); + } + + #[test] + fn youtube_shorts_host() { + assert_eq!( + normalize_http_identity_url("https://youtube.com/shorts/AbCdEfGhIjK").as_deref(), + Some("https://www.youtube.com/shorts/AbCdEfGhIjK") + ); + } + + #[test] + fn arbitrary_query_sorted() { + assert_eq!( + normalize_http_identity_url("https://example.com/x?z=1&a=2").as_deref(), + Some("https://example.com/x?a=2&z=1") + ); + } +} Side B — contributor: tommy-mor Side B — commit message: [5db58b98] Improve vote pair selection for spanning trees and rank refinement. Prefer attaching unranked items to established components before comparing isolates, then zip down adjacent rank-centrality pairs once the pool is fully connected, skipping pairs that already have votes. Co-authored-by: Cursor Side B — unified diff (full patch): diff --git a/server/src/pair.rs b/server/src/pair.rs index 54b5d2417e9dba04ed8df422156e274c2b2f76b2..c14de4b0502c8b5a17cddf3746077739d56e03e0 100644 --- a/server/src/pair.rs +++ b/server/src/pair.rs @@ -3,13 +3,21 @@ //! Pair selection prefers **bridge** votes — comparisons between items in //! different connected components of the voted-pairs graph — so the pool //! merges into one ranking group before refining within it. +//! +//! Among unvoted bridges, prefer merging established voted components, then +//! attaching a never-voted child to an established component, and only then +//! comparing two never-voted children (so the voted graph grows as one tree). +//! +//! Once every pool child sits in one voted component, refinement **zips** down +//! the rank-centrality order: prefer 1 vs 2, then 2 vs 3, and so on, skipping +//! pairs that already have a vote. use rand::seq::SliceRandom; use std::collections::{HashMap, HashSet}; use crate::{ path_types::ItemId, - ranking::connected_components_from_voted_pairs, + ranking::{connected_components_from_voted_pairs, ranked_items}, reducer::{GlobalTree, GroupState}, }; @@ -28,36 +36,77 @@ fn pair_is_voted(group: &GroupState, a: &ItemId, b: &ItemId) -> bool { group.voted_pairs.contains(&(i, j)) } -/// Component id per pool item: voted-pairs graph components plus one id per -/// never-voted child. -fn component_ids(group: &GroupState, pool: &[ItemId]) -> HashMap { +/// Voted-pairs layout for pool items: component id per item plus which ids are +/// multi-node voted components (ranked groups in the UI). +struct ComponentLayout { + ids: HashMap, + established: HashSet, +} + +fn component_layout(group: &GroupState, pool: &[ItemId]) -> ComponentLayout { let n = group.idx_to_item.len(); let (comps, isolates) = connected_components_from_voted_pairs(n, group.voted_pairs.iter().copied()); - let mut out: HashMap = HashMap::new(); + let mut established = HashSet::new(); + let mut ids: HashMap = HashMap::new(); for (comp_idx, comp) in comps.iter().enumerate() { + if comp.len() >= 2 { + established.insert(comp_idx); + } for &idx in comp { if idx < n { - out.insert(group.idx_to_item[idx].clone(), comp_idx); + ids.insert(group.idx_to_item[idx].clone(), comp_idx); } } } let mut next = comps.len(); for &idx in &isolates { if idx < n { - out.insert(group.idx_to_item[idx].clone(), next); + ids.insert(group.idx_to_item[idx].clone(), next); next += 1; } } for item in pool { - out.entry(item.clone()).or_insert_with(|| { + ids.entry(item.clone()).or_insert_with(|| { let id = next; next += 1; id }); } - out + ComponentLayout { ids, established } +} + +/// Every pool child shares one multi-node voted component (spanning tree phase done). +fn pool_fully_connected(layout: &ComponentLayout, pool: &[ItemId]) -> bool { + if pool.len() < 2 { + return false; + } + let mut comp_id = None; + for item in pool { + let Some(id) = layout.ids.get(item) else { + return false; + }; + if !layout.established.contains(id) { + return false; + } + match comp_id { + None => comp_id = Some(*id), + Some(expected) if expected == *id => {} + _ => return false, + } + } + comp_id.is_some() +} + +/// Pool children that appear in `group`, sorted best rank first. +fn ranked_pool_order(group: &GroupState, pool: &[ItemId]) -> Vec { + let pool_set: HashSet<_> = pool.iter().collect(); + ranked_items(group) + .into_iter() + .map(|r| r.item) + .filter(|id| pool_set.contains(id)) + .collect() } #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] @@ -72,19 +121,108 @@ enum PairPriority { WithinVoted = 3, } -fn pair_priority( +/// Tie-break among unvoted bridge pairs. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +enum BridgeSubPriority { + /// Both endpoints lie in established (multi-node) voted components. + MergeEstablished = 0, + /// One established component member and one never-voted child. + AttachIsolate = 1, + /// Two never-voted children (separate singleton components). + IsolatePair = 2, +} + +/// Tie-break among within-component pairs once the pool is one connected group. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +struct WithinSubPriority { + /// 1 = adjacent ranks (i vs i+1); larger = farther apart in the order. + rank_gap: usize, + /// min rank index of the two — zip from the top (1 vs 2 before 2 vs 3). + zip_index: usize, +} + +const WITHIN_SUB_WORST: WithinSubPriority = WithinSubPriority { + rank_gap: usize::MAX, + zip_index: usize::MAX, +}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +struct PairSortKey { + priority: PairPriority, + bridge_sub: BridgeSubPriority, + within_sub: WithinSubPriority, +} + +fn item_in_established(layout: &ComponentLayout, item: &ItemId) -> bool { + layout + .ids + .get(item) + .is_some_and(|id| layout.established.contains(id)) +} + +fn bridge_sub_priority(layout: &ComponentLayout, a: &ItemId, b: &ItemId) -> BridgeSubPriority { + let a_est = item_in_established(layout, a); + let b_est = item_in_established(layout, b); + match (a_est, b_est) { + (true, true) => BridgeSubPriority::MergeEstablished, + (true, false) | (false, true) => BridgeSubPriority::AttachIsolate, + (false, false) => BridgeSubPriority::IsolatePair, + } +} + +fn within_sub_priority( + group: &GroupState, + pool: &[ItemId], + layout: &ComponentLayout, + a: &ItemId, + b: &ItemId, +) -> WithinSubPriority { + if !pool_fully_connected(layout, pool) { + return WITHIN_SUB_WORST; + } + let order = ranked_pool_order(group, pool); + let (Some(i), Some(j)) = (order.iter().position(|x| x == a), order.iter().position(|x| x == b)) + else { + return WITHIN_SUB_WORST; + }; + WithinSubPriority { + rank_gap: i.abs_diff(j), + zip_index: i.min(j), + } +} + +fn pair_sort_key( group: &GroupState, - components: &HashMap, + pool: &[ItemId], + layout: &ComponentLayout, a: &ItemId, b: &ItemId, -) -> PairPriority { +) -> PairSortKey { let voted = pair_is_voted(group, a, b); - let bridge = components.get(a) != components.get(b); - match (bridge, voted) { + let bridge = layout.ids.get(a) != layout.ids.get(b); + let priority = match (bridge, voted) { (true, false) => PairPriority::BridgeUnvoted, (false, false) => PairPriority::WithinUnvoted, (true, true) => PairPriority::BridgeVoted, (false, true) => PairPriority::WithinVoted, + }; + let bridge_sub = if priority == PairPriority::BridgeUnvoted { + bridge_sub_priority(layout, a, b) + } else { + BridgeSubPriority::MergeEstablished + }; + let within_sub = if matches!( + priority, + PairPriority::WithinUnvoted | PairPriority::WithinVoted + ) { + within_sub_priority(group, pool, layout, a, b) + } else { + WITHIN_SUB_WORST + }; + PairSortKey { + priority, + bridge_sub, + within_sub, } } @@ -109,9 +247,12 @@ fn candidate_pairs(pool: &[ItemId], exclude: Option<(&ItemId, &ItemId)>) -> Vec< /// Pick the next pair to vote on within `pool`. /// -/// 1. Prefer unvoted **bridge** pairs (connect separate ranking components). -/// 2. Then unvoted within-component pairs (refinement). -/// 3. Then already-voted pairs (re-compare). +/// 1. Prefer unvoted **bridge** pairs (connect separate ranking components), +/// with sub-priority: merge established components, attach an isolate to +/// established, then compare two isolates. +/// 2. Then unvoted within-component pairs; when the pool is one connected group, +/// prefer adjacent ranks (1 vs 2, 2 vs 3, …) in order, skipping voted pairs. +/// 3. Then already-voted pairs (re-compare), with the same zip ordering. pub fn suggest_next_pair_in_pool( group: &GroupState, pool: &[ItemId], @@ -121,15 +262,15 @@ pub fn suggest_next_pair_in_pool( if candidates.is_empty() { return None; } - let components = component_ids(group, pool); + let layout = component_layout(group, pool); let best = candidates .iter() - .map(|(a, b)| (pair_priority(group, &components, a, b), (a, b))) - .min_by_key(|(p, _)| *p)? + .map(|(a, b)| (pair_sort_key(group, pool, &layout, a, b), (a, b))) + .min_by_key(|(k, _)| *k)? .0; let best_pairs: Vec<(ItemId, ItemId)> = candidates .into_iter() - .filter(|(a, b)| pair_priority(group, &components, a, b) == best) + .filter(|(a, b)| pair_sort_key(group, pool, &layout, a, b) == best) .collect(); best_pairs.choose(&mut rand::thread_rng()).cloned() } @@ -303,6 +444,38 @@ mod tests { assert!(from_ab && from_cd, "expected bridge pair, got {:?}", chosen); } + #[test] + fn suggest_prefers_attach_over_isolate_pair_among_many_unranked() { + let parent = ItemId::parse("reddit.com/r/rust").unwrap(); + let mut tree = seed_children( + &parent, + &[ + "reddit.com/r/rust/a", + "reddit.com/r/rust/b", + "reddit.com/r/rust/c", + "reddit.com/r/rust/d", + "reddit.com/r/rust/e", + ], + ); + let ab = + VoteData::from_recorded(1, "reddit.com/r/rust/a", "reddit.com/r/rust/b", 2, 1).unwrap(); + tree.apply_vote(&parent, ab); + let group = tree.get(&parent).unwrap().local_ranking.clone(); + let pool = children_of(&tree, &parent); + let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); + let chosen = pair_set(&pair); + let from_ab = + chosen.contains("reddit.com/r/rust/a") || chosen.contains("reddit.com/r/rust/b"); + let from_cde = chosen.contains("reddit.com/r/rust/c") + || chosen.contains("reddit.com/r/rust/d") + || chosen.contains("reddit.com/r/rust/e"); + assert!( + from_ab && from_cde, + "expected ranked+unranked attach, got {:?}", + chosen + ); + } + #[test] fn suggest_connects_isolate_to_existing_component() { let parent = ItemId::parse("reddit.com/r/rust").unwrap(); @@ -325,6 +498,65 @@ mod tests { assert!(chosen.contains("reddit.com/r/rust/a") || chosen.contains("reddit.com/r/rust/b")); } + #[test] + fn suggest_zips_adjacent_ranks_when_tree_complete() { + let parent = ItemId::parse("reddit.com/r/rust").unwrap(); + let mut tree = seed_children( + &parent, + &[ + "reddit.com/r/rust/a", + "reddit.com/r/rust/b", + "reddit.com/r/rust/c", + ], + ); + // Star at a connects all three; b-c is the only unvoted adjacent pair left. + for (a, b, l, r) in [ + ("reddit.com/r/rust/a", "reddit.com/r/rust/b", 3, 1), + ("reddit.com/r/rust/a", "reddit.com/r/rust/c", 2, 1), + ] { + let v = VoteData::from_recorded(1, a, b, l, r).unwrap(); + tree.apply_vote(&parent, v); + } + let group = tree.get(&parent).unwrap().local_ranking.clone(); + let pool = children_of(&tree, &parent); + let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); + let chosen = pair_set(&pair); + // a-b and a-c voted; b-c is the only unvoted adjacent pair in rank order. + assert!(chosen.contains("reddit.com/r/rust/b")); + assert!(chosen.contains("reddit.com/r/rust/c")); + } + + #[test] + fn suggest_zip_prefers_1v2_before_2v3_when_both_unvoted() { + let parent = ItemId::parse("reddit.com/r/rust").unwrap(); + let mut tree = seed_children( + &parent, + &[ + "reddit.com/r/rust/a", + "reddit.com/r/rust/b", + "reddit.com/r/rust/c", + "reddit.com/r/rust/d", + ], + ); + // Hub at c connects all four; leave rank-adjacent a-b and b-c unvoted. + for (a, b, l, r) in [ + ("reddit.com/r/rust/c", "reddit.com/r/rust/d", 3, 1), + ("reddit.com/r/rust/b", "reddit.com/r/rust/c", 2, 1), + ("reddit.com/r/rust/a", "reddit.com/r/rust/c", 2, 1), + ] { + let v = VoteData::from_recorded(1, a, b, l, r).unwrap(); + tree.apply_vote(&parent, v); + } + let group = tree.get(&parent).unwrap().local_ranking.clone(); + let pool = children_of(&tree, &parent); + assert!(pool_fully_connected(&component_layout(&group, &pool), &pool)); + let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); + let chosen = pair_set(&pair); + // Top adjacent unvoted edge should be a-b (zip index 0), not b-c (index 1). + assert!(chosen.contains("reddit.com/r/rust/a")); + assert!(chosen.contains("reddit.com/r/rust/b")); + } + #[test] fn resolve_pair_picks_from_pool() { let parent = ItemId::parse("reddit.com/r/rust").unwrap();