You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [5ca518f6] url refactor Side A — unified diff (full patch): diff --git a/Cargo.lock b/Cargo.lock index 67a09a3b54f778fa7e857fdd589c3ed9c92e1322..ad7e4fe6d4ba2f2b033916194c1ef1ed873f1d46 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1757,6 +1757,7 @@ name = "slug-types" version = "0.1.0" dependencies = [ "serde", + "url", ] [[package]] @@ -2272,6 +2273,7 @@ dependencies = [ "idna", "percent-encoding", "serde", + "serde_derive", ] [[package]] diff --git a/server/src/api/ui_html.rs b/server/src/api/ui_html.rs index 606f7d6f97a4428efb90d1e0d544934c861fc4c5..cd3e0f0afd972d9ad9e7e4b92c5fa4c22bb8f620 100644 --- a/server/src/api/ui_html.rs +++ b/server/src/api/ui_html.rs @@ -270,10 +270,10 @@ fn post_redirect_location(room: &str, thread_tag: &str) -> String { format!("/t/{tag}") } else { let room = room.trim(); - let Some((a, b)) = room.split_once('/') else { + let Some(seg) = slug_types::room_route_segment(room) else { return "/".to_string(); }; - format!("/r/{a}/{b}/t/{tag}") + format!("/r/{seg}/t/{tag}") } } diff --git a/server/src/api/write_actor.rs b/server/src/api/write_actor.rs index cb78d2f3f95b1bc163c1fb064d1d5f657e18000f..f9c3b8bd3fbf8fcb9c035e1a1572fef0b08fa8a9 100644 --- a/server/src/api/write_actor.rs +++ b/server/src/api/write_actor.rs @@ -19,13 +19,15 @@ use crate::{ use super::auth::{issue_token_for_user, verify_token}; use super::helpers::{now_ms, resolve_item}; use super::validate::{normalize_room_and_thread, validate_ingest_document}; -use slug_types::RpcResult; +use slug_types::{room_route_segment, RpcResult, ROOM_SHORT_ID_LEN}; fn gen_short_id() -> String { use rand::Rng; const ALPHABET: &[u8] = b"0123456789abcdefghijklmnopqrstuvwxyz"; let mut rng = rand::thread_rng(); - (0..7).map(|_| ALPHABET[rng.gen_range(0..ALPHABET.len())] as char).collect() + (0..ROOM_SHORT_ID_LEN) + .map(|_| ALPHABET[rng.gen_range(0..ALPHABET.len())] as char) + .collect() } fn parse_capability(s: &str) -> Result { @@ -55,8 +57,8 @@ async fn broadcast_web_refresh(state: &AppState, room_key: &str, thread_id: &str let feed_id = if room_key == "public" { "thread-feed" } else { "room-thread-feed" }; let thread_url = if room_key == "public" { format!("/t/{thread_id}") - } else if let Some((short, slug)) = room_key.split_once('/') { - format!("/r/{short}/{slug}/t/{thread_id}") + } else if let Some(seg) = room_route_segment(room_key) { + format!("/r/{seg}/t/{thread_id}") } else { format!("/t/{thread_id}") }; @@ -78,8 +80,8 @@ async fn broadcast_web_refresh(state: &AppState, room_key: &str, thread_id: &str let js = builder.build(); let mut path_prefixes = vec![if room_key == "public" { "/".to_string() - } else if let Some((short, slug)) = room_key.split_once('/') { - format!("/r/{short}/{slug}") + } else if let Some(seg) = room_route_segment(room_key) { + format!("/r/{seg}") } else { "/".to_string() }]; diff --git a/server/src/html/forum/nav.rs b/server/src/html/forum/nav.rs index 0ee33d91160fc5542817b5e3e9ab4fee1d0e600f..48fe11e46731670874ff8b6b05baa6f09ae0b7e4 100644 --- a/server/src/html/forum/nav.rs +++ b/server/src/html/forum/nav.rs @@ -1,7 +1,8 @@ use crate::canonical_path::canonicalize_item; use crate::reducer::ScopeId; +use slug_types::room_route_segment; -/// URL helpers for public `/t/…` and private room threads `/r/{short}/{slug}/t/…`. +/// URL helpers for public `/t/…` and private room threads `/r/{short}{slug}/t/…`. #[derive(Clone)] pub struct ThreadNav { pub room_wire: String, @@ -22,18 +23,15 @@ impl ThreadNav { } } - /// `room_id` wire form `shortid/slug`. + /// `room_id` wire form `shortid/slug` (HTTP uses [`slug_types::room_route_segment`]). pub(crate) fn from_room_id(room_id: &str) -> Option { - let (short, slug) = room_id.split_once('/')?; - if short.is_empty() || slug.is_empty() { - return None; - } + let room_seg = room_route_segment(room_id)?; Some(Self { room_wire: room_id.to_string(), scope: ScopeId::Room(room_id.to_string()), - room_path: format!("/r/{short}/{slug}"), - thread_path_prefix: format!("/r/{short}/{slug}/t"), - garden_path_prefix: format!("/r/{short}/{slug}/~"), + room_path: format!("/r/{room_seg}"), + thread_path_prefix: format!("/r/{room_seg}/t"), + garden_path_prefix: format!("/r/{room_seg}/~"), }) } diff --git a/server/src/html/forum/post_single.rs b/server/src/html/forum/post_single.rs index c316f8f836df9d4ef9c05ebd9e54f699540e6d72..473747b3da3d4d7a54b5e0c63165d2533df643e1 100644 --- a/server/src/html/forum/post_single.rs +++ b/server/src/html/forum/post_single.rs @@ -93,12 +93,14 @@ pub async fn thread_post_view( pub async fn room_thread_post_view( State(state): State, - Path((room_short, room_slug, tag, index_str)): Path<(String, String, String, String)>, + Path((room_key, tag, index_str)): Path<(String, String, String)>, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let reduced = state.reduced.read().await; let user = optional_principal(&headers, &jar, &reduced); if !user_can_view_room(&reduced, &room_id, user.as_deref()) { diff --git a/server/src/html/forum/views.rs b/server/src/html/forum/views.rs index be5df1745ef580891a167c23c3dd6c06f804f295..1ec421f84335cbfe7f9db775b73a8ed24b197174 100644 --- a/server/src/html/forum/views.rs +++ b/server/src/html/forum/views.rs @@ -183,16 +183,18 @@ pub async fn thread_view( thread_view_inner(state, tag, q, ThreadNav::public(), headers, jar, uri).await } -/// Room thread — `/r/:short/:slug/t/:tag` +/// Room thread — `/r/:room_key/t/:tag` (`room_key` = `{short}{slug}`). pub async fn room_thread_view( State(state): State, - Path((room_short, room_slug, tag)): Path<(String, String, String)>, + Path((room_key, tag)): Path<(String, String)>, Query(q): Query, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let reduced = state.reduced.read().await; let user = optional_principal(&headers, &jar, &reduced); if !user_can_view_room(&reduced, &room_id, user.as_deref()) { @@ -226,15 +228,17 @@ pub(super) fn room_not_found_page(jar: &CookieJar, uri: &Uri) -> impl IntoRespon (StatusCode::NOT_FOUND, Html(page.into_string())) } -/// Private room index — `/r/:short/:slug` +/// Private room index — `/r/:room_key` pub async fn room_page( State(state): State, - Path((room_short, room_slug)): Path<(String, String)>, + Path(room_key): Path, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "room not found").into_response(); + }; let now = now_ms(); let reduced = state.reduced.read().await; if !reduced.rooms.contains(&room_id) { @@ -266,7 +270,10 @@ pub async fn room_page( let audit_cli = format!("npx slugsocial private {room_id} audit"); drop(reduced); - let slug_display = room_slug.as_str(); + let slug_display = room_id + .split_once('/') + .map(|(_, slug)| slug) + .unwrap_or(room_id.as_str()); let page = layout( &format!("room {slug_display} — slug.social"), "view-thread", diff --git a/server/src/html/garden.rs b/server/src/html/garden.rs index 423f23fd8c9ad7b7f454d6ea7a9a7607a4c9c5b9..e615dd356bcf634232d85610c0a26235ead125fd 100644 --- a/server/src/html/garden.rs +++ b/server/src/html/garden.rs @@ -309,12 +309,14 @@ pub async fn external_ontology_path( pub async fn room_garden_index( State(state): State, - Path((room_short, room_slug)): Path<(String, String)>, + Path(room_key): Path, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let Some(nav) = ThreadNav::from_room_id(&room_id) else { return (StatusCode::NOT_FOUND, "bad room path").into_response(); }; @@ -341,12 +343,14 @@ pub async fn room_garden_index( pub async fn room_external_garden_index( State(state): State, - Path((room_short, room_slug)): Path<(String, String)>, + Path(room_key): Path, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let Some(nav) = ThreadNav::from_room_id(&room_id) else { return (StatusCode::NOT_FOUND, "bad room path").into_response(); }; @@ -416,12 +420,14 @@ pub async fn room_external_garden_index( pub async fn room_external_ontology_path( State(state): State, - Path((room_short, room_slug, path)): Path<(String, String, String)>, + Path((room_key, path)): Path<(String, String)>, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let Some(nav) = ThreadNav::from_room_id(&room_id) else { return (StatusCode::NOT_FOUND, "bad room path").into_response(); }; @@ -442,12 +448,14 @@ pub async fn room_external_ontology_path( pub async fn room_ontology_path( State(state): State, - Path((room_short, room_slug, path)): Path<(String, String, String)>, + Path((room_key, path)): Path<(String, String)>, headers: HeaderMap, jar: CookieJar, uri: Uri, ) -> impl IntoResponse { - let room_id = format!("{room_short}/{room_slug}"); + let Some(room_id) = slug_types::room_id_from_route_segment(&room_key) else { + return (StatusCode::NOT_FOUND, "bad room path").into_response(); + }; let Some(nav) = ThreadNav::from_room_id(&room_id) else { return (StatusCode::NOT_FOUND, "bad room path").into_response(); }; diff --git a/server/src/html/search.rs b/server/src/html/search.rs index 43e6ebf36cf0fe72c96f0f9d850bea51ac094c43..f01732f7edb32c68fcc10c39545c8f56476adb27 100644 --- a/server/src/html/search.rs +++ b/server/src/html/search.rs @@ -351,8 +351,8 @@ fn render_search_results(results: &SearchResults, query: &str) -> Markup { ul class="search-posts" { @for r in &results.posts { @let (post_href, post_label) = if let Some((room, tag)) = r.thread.split_once("/#") { - if let Some((short, slug)) = room.split_once('/') { - (format!("/r/{short}/{slug}/t/{tag}"), format!("{room}/#{tag}")) + if let Some(seg) = slug_types::room_route_segment(room) { + (format!("/r/{seg}/t/{tag}"), format!("{room}/#{tag}")) } else { ("/".to_string(), r.thread.clone()) } diff --git a/server/src/lib.rs b/server/src/lib.rs index 9e69c3f4a153527caa77b00ad101aca94b425995..71032f9a0b556097d90f3b403c677e0c86429ee4 100644 --- a/server/src/lib.rs +++ b/server/src/lib.rs @@ -79,30 +79,30 @@ pub fn create_app(state: AppState) -> Router { .route("/~/*path", get(crate::html::ontology_path)) .route("/-", get(crate::html::external_garden_index)) .route("/-/*path", get(crate::html::external_ontology_path)) - .route("/r/:room_short/:room_slug/~", get(crate::html::room_garden_index)) + .route("/r/:room_key/~", get(crate::html::room_garden_index)) .route( - "/r/:room_short/:room_slug/~/*path", + "/r/:room_key/~/*path", get(crate::html::room_ontology_path), ) .route( - "/r/:room_short/:room_slug/-", + "/r/:room_key/-", get(crate::html::room_external_garden_index), ) .route( - "/r/:room_short/:room_slug/-/*path", + "/r/:room_key/-/*path", get(crate::html::room_external_ontology_path), ) .route("/t/:tag/:index", get(crate::html::thread_post_view)) .route("/t/:tag", get(crate::html::thread_view)) .route( - "/r/:room_short/:room_slug/t/:thread_tag/:index", + "/r/:room_key/t/:thread_tag/:index", get(crate::html::room_thread_post_view), ) .route( - "/r/:room_short/:room_slug/t/:thread_tag", + "/r/:room_key/t/:thread_tag", get(crate::html::room_thread_view), ) - .route("/r/:room_short/:room_slug", get(crate::html::room_page)) + .route("/r/:room_key", get(crate::html::room_page)) .route("/join/:token", get(api::get_join_invite)) .route("/auth/login", get(api::get_auth_login)) .route("/auth/callback", get(api::get_auth_callback)) diff --git a/server/tests/integration.rs b/server/tests/integration.rs index 515c691d04ef3f1f4bdfa8ee911360a43fd2b26e..feb73d475b3fc635d7f16c329d4600718c7472ba 100644 --- a/server/tests/integration.rs +++ b/server/tests/integration.rs @@ -1,4 +1,5 @@ use sha2::{Digest, Sha256}; +use slug_types::room_route_segment; use slugsocial_server::{ event_log::EventLog, events::{Event, TokenIssued, UserRegistered}, @@ -607,7 +608,7 @@ async fn test_private_room_thread_urls_use_t_segment() { .as_str() .unwrap() .to_string(); - let (room_short, room_slug) = room_id.split_once('/').unwrap(); + let room_seg = room_route_segment(&room_id).unwrap(); let rpc = ui_post_ingest_rpc(&room_id, "main-thread", "private post via web"); let post = client @@ -626,7 +627,7 @@ async fn test_private_room_thread_urls_use_t_segment() { Some("text/javascript; charset=utf-8") ); let post_js = post.text().await.unwrap(); - let location = format!("/r/{room_short}/{room_slug}/t/main-thread"); + let location = format!("/r/{room_seg}/t/main-thread"); assert!(post_js.contains(&format!("window.location = {:?};", location))); assert!(post_js.contains("#room-thread-feed")); assert!(post_js.contains("#thread-feed-region")); @@ -665,7 +666,7 @@ async fn test_private_room_post_links_use_private_garden_routes() { .as_str() .unwrap() .to_string(); - let (room_short, room_slug) = room_id.split_once('/').unwrap(); + let room_seg = room_route_segment(&room_id).unwrap(); let rpc = ui_post_ingest_rpc( &room_id, @@ -682,18 +683,18 @@ async fn test_private_room_post_links_use_private_garden_routes() { assert_eq!(post.status(), reqwest::StatusCode::OK); let thread_page = client - .get(format!("http://{addr}/r/{room_short}/{room_slug}/t/garden-thread")) + .get(format!("http://{addr}/r/{room_seg}/t/garden-thread")) .header("Authorization", format!("Bearer {bearer}")) .send() .await .unwrap(); assert!(thread_page.status().is_success()); let body = thread_page.text().await.unwrap(); - assert!(body.contains(&format!("/r/{room_short}/{room_slug}/~/secret/item"))); + assert!(body.contains(&format!("/r/{room_seg}/~/secret/item"))); assert!(!body.contains("href=\"/~/secret/item\"")); let garden_page = client - .get(format!("http://{addr}/r/{room_short}/{room_slug}/~/secret/item")) + .get(format!("http://{addr}/r/{room_seg}/~/secret/item")) .header("Authorization", format!("Bearer {bearer}")) .send() .await @@ -701,7 +702,7 @@ async fn test_private_room_post_links_use_private_garden_routes() { assert!(garden_page.status().is_success()); let garden_body = garden_page.text().await.unwrap(); assert!(garden_body.contains("classified")); - assert!(garden_body.contains(&format!("/r/{room_short}/{room_slug}/t/garden-thread"))); + assert!(garden_body.contains(&format!("/r/{room_seg}/t/garden-thread"))); } #[tokio::test] @@ -726,7 +727,7 @@ async fn test_private_room_garden_root_lists_top_level_tilde_children() { .as_str() .unwrap() .to_string(); - let (room_short, room_slug) = room_id.split_once('/').unwrap(); + let room_seg = room_route_segment(&room_id).unwrap(); let rpc = ui_post_ingest_rpc( &room_id, @@ -743,7 +744,7 @@ async fn test_private_room_garden_root_lists_top_level_tilde_children() { assert_eq!(post.status(), reqwest::StatusCode::OK); let root_page = client - .get(format!("http://{addr}/r/{room_short}/{room_slug}/~")) + .get(format!("http://{addr}/r/{room_seg}/~")) .header("Authorization", format!("Bearer {bearer}")) .send() .await @@ -782,10 +783,10 @@ async fn test_empty_private_room_garden_returns_404() { .as_str() .unwrap() .to_string(); - let (room_short, room_slug) = room_id.split_once('/').unwrap(); + let room_seg = room_route_segment(&room_id).unwrap(); let root = client - .get(format!("http://{addr}/r/{room_short}/{room_slug}/~")) + .get(format!("http://{addr}/r/{room_seg}/~")) .header("Authorization", format!("Bearer {bearer}")) .send() .await @@ -896,8 +897,8 @@ async fn test_sse_stream_emits_evalable_js_after_post() { .as_str() .unwrap() .to_string(); - let (room_short, room_slug) = room_id.split_once('/').unwrap(); - let room_path = format!("/r/{room_short}/{room_slug}"); + let room_seg = room_route_segment(&room_id).unwrap(); + let room_path = format!("/r/{room_seg}"); let sse_resp = client .get(format!("http://{addr}/sse?path={}", urlencoding::encode(&room_path))) diff --git a/test/browser_room_delete.clj b/test/browser_room_delete.clj index 833e3f8aa7f850b60ee59742736143cbf101a07d..764ecaa674798a61c1ffca3859f2a56a91a0e97e 100644 --- a/test/browser_room_delete.clj +++ b/test/browser_room_delete.clj @@ -54,7 +54,7 @@ room-id (get-in create-json ["results" 0 "result" "RoomCreated" "room_id"]) _ (is (string? room-id) "room id present") [room-short room-slug] (str/split room-id #"/" 2) - room-path (str "/r/" room-short "/" room-slug)] + room-path (str "/r/" room-short room-slug)] (core/with-playwright [pw] (core/with-browser [browser (core/launch-chromium pw {:headless true :channel "chrome"})] (core/with-context [ctx (core/new-context browser)] diff --git a/test/browser_sse.clj b/test/browser_sse.clj index 8bb064dbc2dc0feb6caaa67f520a461645958277..678399b79f78ff48b5e77342fa21e69992393834 100644 --- a/test/browser_sse.clj +++ b/test/browser_sse.clj @@ -77,7 +77,7 @@ (login-user! bob-pg base-url "bob") (let [[room-short room-slug] (str/split room-id #"/" 2) - room-url (str base-url "/r/" room-short "/" room-slug) + room-url (str base-url "/r/" room-short room-slug) thread-url (str room-url "/t/sse-thread")] ;; Object under test: slug_ui.js intercepts POST /ui, evals JS, morphs ;; #new-thread-ui-slot (expand compose), then post_ingest redirects to thread. diff --git a/test/walkthrough_fixture.clj b/test/walkthrough_fixture.clj index a487ff1ef8548580577914fb33f321586859c9af..f7eb1cacca0317129e9baf68fa4f4e33356ad70f 100644 --- a/test/walkthrough_fixture.clj +++ b/test/walkthrough_fixture.clj @@ -88,9 +88,9 @@ :room {:id room-id :short room-short :slug room-slug - :url (str base-url "/r/" room-short "/" room-slug) - :thread_url (str base-url "/r/" room-short "/" room-slug "/t/walkthrough-thread") - :garden_url (str base-url "/r/" room-short "/" room-slug "/~/secret/item")}})) + :url (str base-url "/r/" room-short room-slug) + :thread_url (str base-url "/r/" room-short room-slug "/t/walkthrough-thread") + :garden_url (str base-url "/r/" room-short room-slug "/~/secret/item")}})) (defn- rebase-fixture-summary [saved current-base-url current-google-url current-data-dir] (let [inner (:summary saved) @@ -103,9 +103,9 @@ :data_dir (str current-data-dir) :summary (assoc inner :room (assoc room - :url (str current-base-url "/r/" rs "/" lg) - :thread_url (str current-base-url "/r/" rs "/" lg "/t/walkthrough-thread") - :garden_url (str current-base-url "/r/" rs "/" lg "/~/secret/item")))))) + :url (str current-base-url "/r/" rs lg) + :thread_url (str current-base-url "/r/" rs lg "/t/walkthrough-thread") + :garden_url (str current-base-url "/r/" rs lg "/~/secret/item")))))) (defn- fixture-log-present? [data-dir] (let [p (fs/path data-dir "events.jsonl")] diff --git a/types/Cargo.toml b/types/Cargo.toml index 1ed06d483ac7c126516bce27f90e7639c2dfe370..5dc3becf238f23243313dff0484c682a7b225737 100644 --- a/types/Cargo.toml +++ b/types/Cargo.toml @@ -5,3 +5,4 @@ edition = "2021" [dependencies] serde = { version = "1.0", features = ["derive"] } +url = { version = "2.5", features = ["serde"] } diff --git a/types/src/lib.rs b/types/src/lib.rs index fce1b5fa5e259a4b86a11cb4d4f7cc9bec3a0aa1..516cf935f15fca97081b39b988da5be894c67725 100644 --- a/types/src/lib.rs +++ b/types/src/lib.rs @@ -1,5 +1,7 @@ use serde::{Deserialize, Serialize}; +pub mod room_route; +pub mod url_normalize; pub mod paths; pub mod timeago; @@ -8,6 +10,8 @@ pub use paths::{ CanonicalItemUrl, ForumThreadUrl, GardenItemUrl, RelativePath, SLUG_TILDE_ONTOLOGY_ROOT, TildeHttpPathTail, TildeOntologyPath, TildePath, tilde_http_path_to_canonical, }; +pub use room_route::{room_id_from_route_segment, room_route_segment, ROOM_SHORT_ID_LEN}; +pub use url_normalize::normalize_http_identity_url; /// Max characters returned for a garden item body unless `full=true` / `--full` (API + CLI). pub const MAX_ITEM_BODY_PREVIEW_CHARS: usize = 100_000; @@ -657,3 +661,6 @@ pub struct VoteResponse { pub ranking: Vec, pub next: NextMoves, } + +#[cfg(test)] +mod url_identity_tests; diff --git a/types/src/paths.rs b/types/src/paths.rs index 8971f599a7ebcf399f60c306d35ee28d0f581194..98787a5fb481a1599556564bbbbd1d54dc693fc3 100644 --- a/types/src/paths.rs +++ b/types/src/paths.rs @@ -5,7 +5,7 @@ //! //! - **[`canonicalize_item`] / [`CanonicalItemUrl`]** — graph storage key and DSL form; tilde //! ontology root is always [`SLUG_TILDE_ONTOLOGY_ROOT`] (no `…/~/` trailing slash only). -//! - **[`TildeHttpPathTail`]** — capture from `GET /~/*path` or `…/r/…/~/…` (the `*path` segment). +//! - **[`TildeHttpPathTail`]** — capture from `GET /~/*path` or `…/r/{short}{slug}/~/…` (the `*path` segment). //! - **`-/…` wire form** — external items; see [`canonicalize_item`] dash branch. //! - **[`GardenItemUrl`], [`ForumThreadUrl`]** — JSON / browser href surfaces. @@ -15,6 +15,9 @@ use std::ops::Deref; use serde::{Deserialize, Serialize}; +use crate::room_route::room_route_segment; +use crate::url_normalize::{host_preserves_dash_path_case, normalize_http_identity_url}; + // --------------------------------------------------------------------------- // Slug tilde ontology (single storage form for `~/`) // --------------------------------------------------------------------------- @@ -41,6 +44,29 @@ pub fn canonicalize_tag(input: &str) -> String { input.trim().trim_start_matches('#').to_lowercase() } +fn finalize_external_identity_url(s: String) -> String { + if s.starts_with("https://slug.social/") { + return s; + } + let normalized = normalize_http_identity_url(&s).unwrap_or_else(|| s.clone()); + strip_redundant_root_slash(&normalized).unwrap_or(normalized) +} + +/// `url::Url` serializes bare hosts with a `/` path; we keep host-only items slash-free for stable +/// keys matching the pre-normalizer spellings. +fn strip_redundant_root_slash(s: &str) -> Option { + let u = url::Url::parse(s).ok()?; + if u.path() == "/" && u.query().is_none() && u.fragment().is_none() { + let scheme = u.scheme(); + let host = u.host_str()?; + return Some(match u.port() { + Some(p) => format!("{scheme}://{host}:{p}"), + None => format!("{scheme}://{host}"), + }); + } + None +} + /// Ontology item reference → canonical absolute URL on the slug host. pub fn canonicalize_item(input: &str) -> String { let s = input.trim(); @@ -57,8 +83,9 @@ pub fn canonicalize_item(input: &str) -> String { if host.is_empty() { return String::new(); } + let preserve_case = host_preserves_dash_path_case(&host); return if tail.is_empty() { - format!("https://{}", host) + finalize_external_identity_url(format!("https://{}", host)) } else { let path = tail .trim_start_matches('/') @@ -68,33 +95,35 @@ pub fn canonicalize_item(input: &str) -> String { let t = seg.trim(); if t.is_empty() { None + } else if preserve_case { + Some(t.to_string()) } else { Some(t.to_lowercase()) } }) .collect::>() .join("/"); - format!("https://{}/{}", host, path) + finalize_external_identity_url(format!("https://{}/{}", host, path)) }; } if let Some(rest) = s.strip_prefix("https://") { let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); let host = host.trim().to_lowercase(); - if tail.is_empty() { - return format!("https://{}", host); + return finalize_external_identity_url(if tail.is_empty() { + format!("https://{}", host) } else { - return format!("https://{}/{}", host, tail); - } + format!("https://{}/{}", host, tail) + }); } if let Some(rest) = s.strip_prefix("http://") { let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); let host = host.trim().to_lowercase(); - if tail.is_empty() { - return format!("http://{}", host); + return finalize_external_identity_url(if tail.is_empty() { + format!("http://{}", host) } else { - return format!("http://{}/{}", host, tail); - } + format!("http://{}/{}", host, tail) + }); } let is_tilde = s.starts_with("~/"); @@ -319,7 +348,7 @@ impl CanonicalItemUrl { } } -/// HTTP route capture: path segment after `~/` in `GET /~/*path` or `…/r/…/~/…` (empty = ontology root). +/// HTTP route capture: path segment after `~/` in `GET /~/*path` or `…/r/{short}{slug}/~/…` (empty = ontology root). #[derive(Debug, Clone, PartialEq, Eq, Hash)] pub struct TildeHttpPathTail(pub String); @@ -506,12 +535,9 @@ fn garden_href_string(item: &str, room_wire: &str) -> String { if room.is_empty() || room == "public" { return api_path_or_url(item); } - let Some((short, slug)) = room.split_once('/') else { + let Some(room_seg) = room_route_segment(room) else { return api_path_or_url(item); }; - if short.is_empty() || slug.is_empty() { - return api_path_or_url(item); - } let Some(c) = CanonicalItemUrl::parse(item) else { return api_path_or_url(item); }; @@ -520,19 +546,19 @@ fn garden_href_string(item: &str, room_wire: &str) -> String { let root_norm = root.as_str().trim_end_matches('/'); if let Some(tail) = c.tilde_tail() { return if tail.is_empty() { - format!("https://slug.social/r/{short}/{slug}/~") + format!("https://slug.social/r/{room_seg}/~") } else { - format!("https://slug.social/r/{short}/{slug}/~/{}", tail) + format!("https://slug.social/r/{room_seg}/~/{}", tail) }; } if item_norm == root_norm { - return format!("https://slug.social/r/{short}/{slug}/~"); + return format!("https://slug.social/r/{room_seg}/~"); } // External http(s) items use the same `/-/…` namespace under the room garden. if c.as_str().starts_with("https://") || c.as_str().starts_with("http://") { let tail = c.display_path(); let tail = tail.strip_prefix("-/").unwrap_or(tail.as_str()); - return format!("https://slug.social/r/{short}/{slug}/-/{tail}"); + return format!("https://slug.social/r/{room_seg}/-/{tail}"); } api_path_or_url(item) } @@ -556,12 +582,8 @@ impl ForumThreadUrl { let tag = thread_tag.trim().trim_start_matches('#'); Self(if room.is_empty() || room == "public" { format!("https://slug.social/t/{tag}") - } else if let Some((short, slug)) = room.split_once('/') { - if short.is_empty() || slug.is_empty() { - format!("https://slug.social/t/{tag}") - } else { - format!("https://slug.social/r/{short}/{slug}/t/{tag}") - } + } else if let Some(room_seg) = room_route_segment(room) { + format!("https://slug.social/r/{room_seg}/t/{tag}") } else { format!("https://slug.social/t/{tag}") }) @@ -706,7 +728,7 @@ mod tests { fn garden_private_room_prefixes_ontology() { assert_eq!( GardenItemUrl::from_storage_str("https://slug.social/~/topic/x", "9ab12cd/my-room").as_str(), - "https://slug.social/r/9ab12cd/my-room/~/topic/x" + "https://slug.social/r/9ab12cdmy-room/~/topic/x" ); } @@ -714,11 +736,11 @@ mod tests { fn garden_private_room_ontology_root() { assert_eq!( GardenItemUrl::from_storage_str("https://slug.social/~", "9ab12cd/my-room").as_str(), - "https://slug.social/r/9ab12cd/my-room/~" + "https://slug.social/r/9ab12cdmy-room/~" ); assert_eq!( GardenItemUrl::from_storage_str("https://slug.social/~/", "9ab12cd/my-room").as_str(), - "https://slug.social/r/9ab12cd/my-room/~" + "https://slug.social/r/9ab12cdmy-room/~" ); } @@ -727,7 +749,7 @@ mod tests { let u = "https://example.com/z"; assert_eq!( GardenItemUrl::from_storage_str(u, "9ab12cd/my-room").as_str(), - "https://slug.social/r/9ab12cd/my-room/-/example.com/z" + "https://slug.social/r/9ab12cdmy-room/-/example.com/z" ); } @@ -739,7 +761,19 @@ mod tests { ); assert_eq!( ForumThreadUrl::from_room_tag("9ab12cd/my-room", "#debate").as_str(), - "https://slug.social/r/9ab12cd/my-room/t/debate" + "https://slug.social/r/9ab12cdmy-room/t/debate" + ); + } + + #[test] + fn canonicalize_youtube_short_links() { + assert_eq!( + canonicalize_item("https://youtu.be/dQw4w9WgXcQ"), + "https://www.youtube.com/watch?v=dQw4w9WgXcQ" + ); + assert_eq!( + canonicalize_item("-/youtu.be/dQw4w9WgXcQ"), + "https://www.youtube.com/watch?v=dQw4w9WgXcQ" ); } diff --git a/types/src/room_route.rs b/types/src/room_route.rs new file mode 100644 index 0000000000000000000000000000000000000000..4f4780c88e30f2b28e2cfcd7aee713d39dbd1c66 --- /dev/null +++ b/types/src/room_route.rs @@ -0,0 +1,56 @@ +//! HTTP path encoding for private rooms: `/r/{short}{slug}` (short is fixed width). + +/// Byte length of the random `short` segment in `short/slug` room ids. +/// Must match room creation (`gen_short_id`) and [`super::paths`][] URL builders. +pub const ROOM_SHORT_ID_LEN: usize = 7; + +/// `ab12cde/my-room` → `ab12cdemy-room` for a single `/r/…` path segment. +pub fn room_route_segment(room_id: &str) -> Option { + let (short, slug) = room_id.split_once('/')?; + if short.len() != ROOM_SHORT_ID_LEN || short.is_empty() || slug.is_empty() { + return None; + } + if !short + .bytes() + .all(|b| matches!(b, b'0'..=b'9' | b'a'..=b'z')) + { + return None; + } + Some(format!("{short}{slug}")) +} + +/// `/r/{short}{slug}` path segment → `short/slug` wire id (inverse of [`room_route_segment`]). +pub fn room_id_from_route_segment(seg: &str) -> Option { + if seg.len() <= ROOM_SHORT_ID_LEN { + return None; + } + let (short, slug) = seg.split_at(ROOM_SHORT_ID_LEN); + if short.is_empty() || slug.is_empty() { + return None; + } + if !short + .bytes() + .all(|b| matches!(b, b'0'..=b'9' | b'a'..=b'z')) + { + return None; + } + Some(format!("{short}/{slug}")) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn round_trip_room_segment() { + let id = "9ab12cd/my-room"; + let seg = room_route_segment(id).unwrap(); + assert_eq!(seg, "9ab12cdmy-room"); + assert_eq!(room_id_from_route_segment(&seg).as_deref(), Some(id)); + } + + #[test] + fn too_short_segment_rejected() { + assert!(room_id_from_route_segment("9ab12cd").is_none()); + } +} diff --git a/types/src/url_identity_tests.rs b/types/src/url_identity_tests.rs new file mode 100644 index 0000000000000000000000000000000000000000..e2ff0a9a64a88049e31f20979c703777df6a9b65 --- /dev/null +++ b/types/src/url_identity_tests.rs @@ -0,0 +1,126 @@ +//! How `url::Url` behaves as `HashMap` keys (`Eq` + `Hash`). +//! +//! If `ItemId::External` stores `Url`, these tests are the contract you are buying into +//! (or the baseline before you add a custom normalization layer). + +use std::collections::HashMap; +use std::hash::{Hash, Hasher}; +use url::Url; + +fn hash_one(url: &Url) -> u64 { + let mut h = std::collections::hash_map::DefaultHasher::new(); + url.hash(&mut h); + h.finish() +} + +#[test] +fn identical_parse_strings_are_eq_and_share_hash_bucket() { + let a = Url::parse("https://example.com/path").unwrap(); + let b = Url::parse("https://example.com/path").unwrap(); + assert_eq!(a, b); + assert_eq!(hash_one(&a), hash_one(&b)); + + let mut m: HashMap = HashMap::new(); + m.insert(a, 1); + *m.entry(b).or_default() += 10; + assert_eq!(m.len(), 1); + assert_eq!(m[&Url::parse("https://example.com/path").unwrap()], 11); +} + +#[test] +fn host_is_ascii_lowercase_in_eq() { + let lower = Url::parse("https://examplE.com/").unwrap(); + let upper = Url::parse("https://EXAMPLE.com/").unwrap(); + assert_eq!(lower, upper); + assert_eq!(hash_one(&lower), hash_one(&upper)); +} + +#[test] +fn path_space_normalizes_to_percent_encoding_so_forms_merge() { + let encoded = Url::parse("https://example.com/a%20b").unwrap(); + let decoded = Url::parse("https://example.com/a b").unwrap(); + // Parser normalizes both to the same internal path (`/a%20b`). + assert_eq!(encoded, decoded); + assert_eq!(hash_one(&encoded), hash_one(&decoded)); + + let mut m: HashMap = HashMap::new(); + m.insert(encoded, "first"); + assert_eq!(m.insert(decoded, "second"), Some("first")); + assert_eq!(m.len(), 1); + assert_eq!(m.values().next().copied(), Some("second")); +} + +#[test] +fn encoded_slash_in_segment_stays_distinct_from_real_path_separator() { + let encoded = Url::parse("https://example.com/a%2Fb").unwrap(); + let real_slash = Url::parse("https://example.com/a/b").unwrap(); + assert_ne!(encoded, real_slash); + assert_ne!(hash_one(&encoded), hash_one(&real_slash)); +} + +#[test] +fn trailing_slash_on_path_is_significant_for_eq() { + let with_slash = Url::parse("https://example.com/foo/").unwrap(); + let no_slash = Url::parse("https://example.com/foo").unwrap(); + assert_ne!(with_slash, no_slash); + assert_ne!(hash_one(&with_slash), hash_one(&no_slash)); +} + +#[test] +fn default_http_port_80_is_normalized_in_representation() { + let explicit = Url::parse("http://example.com:80/").unwrap(); + let implicit = Url::parse("http://example.com/").unwrap(); + assert_eq!(explicit, implicit); + assert_eq!(hash_one(&explicit), hash_one(&implicit)); +} + +#[test] +fn default_https_port_443_is_normalized() { + let explicit = Url::parse("https://example.com:443/foo").unwrap(); + let implicit = Url::parse("https://example.com/foo").unwrap(); + assert_eq!(explicit, implicit); +} + +#[test] +fn non_default_port_is_part_of_identity() { + let a = Url::parse("https://example.com:444/").unwrap(); + let b = Url::parse("https://example.com:445/").unwrap(); + assert_ne!(a, b); +} + +#[test] +fn empty_path_vs_slash_only_path_may_differ() { + let root = Url::parse("https://example.com").unwrap(); + let slash = Url::parse("https://example.com/").unwrap(); + // Both serialize to `https://example.com/` in practice for this crate — verify. + assert_eq!(root, slash, "document: root and trailing-slash-only merge for this parser"); +} + +#[test] +fn scheme_case_is_normalized_to_lowercase() { + let lower = Url::parse("https://example.com/").unwrap(); + let upper = Url::parse("HTTPS://example.com/").unwrap(); + assert_eq!(lower, upper); +} + +#[test] +fn fragment_is_part_of_eq_and_hash() { + let no_frag = Url::parse("https://example.com/a").unwrap(); + let frag = Url::parse("https://example.com/a#section").unwrap(); + assert_ne!( + no_frag, frag, + "#fragment is included in PartialEq — anchors are different HashMap keys" + ); + assert_ne!(hash_one(&no_frag), hash_one(&frag)); +} + +#[test] +fn query_order_and_encoding_can_split_identity() { + let a = Url::parse("https://example.com/?b=2&a=1").unwrap(); + let b = Url::parse("https://example.com/?a=1&b=2").unwrap(); + assert_ne!(a, b, "query pairs order is preserved in serialization"); + + let plus = Url::parse("https://example.com/?q=a+b").unwrap(); + let encoded = Url::parse("https://example.com/?q=a%20b").unwrap(); + assert_ne!(plus, encoded, "space as + vs %20 — different keys unless normalized"); +} diff --git a/types/src/url_normalize.rs b/types/src/url_normalize.rs new file mode 100644 index 0000000000000000000000000000000000000000..d5bc913f33ad29cd0cd1c7158e28e4fefdbf5f5d --- /dev/null +++ b/types/src/url_normalize.rs @@ -0,0 +1,234 @@ +//! Normalization for external `http(s)://` item identity (not slug tilde ontology). +//! +//! Policy (intentional, extend here as new domains need treatment): +//! - Query pairs sorted lexicographically by **lowercased** key, then value. +//! - YouTube family → stable `www.youtube.com` shapes where possible. + +use url::Url; + +/// Whether `-/` → `https://…` path segments should keep original casing (YouTube video IDs are +/// case-sensitive). +pub(crate) fn host_preserves_dash_path_case(host: &str) -> bool { + let h = host.trim().to_ascii_lowercase(); + let h = h.strip_prefix("www.").unwrap_or(h.as_str()); + matches!( + h, + "youtu.be" | "youtube.com" | "m.youtube.com" | "music.youtube.com" + ) +} + +/// Normalize external http(s) URLs for stable [`super::paths::canonicalize_item`] output. +pub fn normalize_http_identity_url(s: &str) -> Option { + let mut u = Url::parse(s).ok()?; + if !matches!(u.scheme(), "http" | "https") { + return None; + } + rewrite_youtube(&mut u); + sort_query_pairs(&mut u); + Some(u.to_string()) +} + +fn base_host(host: &str) -> String { + let lower = host.to_ascii_lowercase(); + lower + .strip_prefix("www.") + .unwrap_or(lower.as_str()) + .to_string() +} + +fn rewrite_youtube(u: &mut Url) { + let Some(host_raw) = u.host_str() else { + return; + }; + let base = base_host(host_raw); + let path = u.path().to_string(); + + match base.as_str() { + "youtu.be" => { + let id = path.trim_start_matches('/').split('/').next().unwrap_or("").to_string(); + if id.is_empty() { + return; + } + let saved: Vec<(String, String)> = u.query_pairs().into_owned().collect(); + let Ok(mut out) = Url::parse(&format!("https://www.youtube.com/watch?v={id}")) else { + return; + }; + { + let mut q = out.query_pairs_mut(); + for (k, v) in saved { + if k.eq_ignore_ascii_case("v") { + continue; + } + q.append_pair(&k, &v); + } + } + *u = out; + } + "youtube.com" | "m.youtube.com" => { + if base == "m.youtube.com" { + let _ = u.set_host(Some("www.youtube.com")); + } + if path.starts_with("/embed/") { + let id = path + .strip_prefix("/embed/") + .unwrap_or("") + .trim_matches('/') + .split('/') + .next() + .unwrap_or("") + .to_string(); + if id.is_empty() { + return; + } + let saved: Vec<(String, String)> = u.query_pairs().into_owned().collect(); + let Ok(mut out) = Url::parse(&format!("https://www.youtube.com/watch?v={id}")) + else { + return; + }; + { + let mut q = out.query_pairs_mut(); + for (k, v) in saved { + if k.eq_ignore_ascii_case("v") { + continue; + } + q.append_pair(&k, &v); + } + } + *u = out; + return; + } + if path.starts_with("/v/") { + let id = path + .strip_prefix("/v/") + .unwrap_or("") + .trim_matches('/') + .split('/') + .next() + .unwrap_or("") + .to_string(); + if id.is_empty() { + return; + } + let saved: Vec<(String, String)> = u.query_pairs().into_owned().collect(); + let Ok(mut out) = Url::parse(&format!("https://www.youtube.com/watch?v={id}")) + else { + return; + }; + { + let mut q = out.query_pairs_mut(); + for (k, v) in saved { + if k.eq_ignore_ascii_case("v") { + continue; + } + q.append_pair(&k, &v); + } + } + *u = out; + return; + } + if path.starts_with("/watch") { + let _ = u.set_host(Some("www.youtube.com")); + return; + } + if path.starts_with("/shorts/") { + let id = path + .strip_prefix("/shorts/") + .unwrap_or("") + .trim_matches('/') + .split('/') + .next() + .unwrap_or("") + .to_string(); + if id.is_empty() { + return; + } + let saved: Vec<(String, String)> = u.query_pairs().into_owned().collect(); + let Ok(mut out) = Url::parse(&format!("https://www.youtube.com/shorts/{id}")) + else { + return; + }; + { + let mut q = out.query_pairs_mut(); + for (k, v) in saved { + q.append_pair(&k, &v); + } + } + *u = out; + return; + } + let _ = u.set_host(Some("www.youtube.com")); + } + "music.youtube.com" => {} + _ => {} + } +} + +fn sort_query_pairs(u: &mut Url) { + let pairs: Vec<(String, String)> = u.query_pairs().into_owned().collect(); + if pairs.is_empty() { + u.set_query(None); + return; + } + let mut pairs = pairs; + pairs.sort_by(|a, b| { + a.0.to_ascii_lowercase() + .cmp(&b.0.to_ascii_lowercase()) + .then_with(|| a.1.cmp(&b.1)) + }); + u.set_query(None); + { + let mut q = u.query_pairs_mut(); + for (k, v) in pairs { + q.append_pair(&k, &v); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn youtube_youtu_be_to_watch() { + assert_eq!( + normalize_http_identity_url("https://youtu.be/dQw4w9WgXcQ").as_deref(), + Some("https://www.youtube.com/watch?v=dQw4w9WgXcQ") + ); + } + + #[test] + fn youtube_watch_query_sorted() { + assert_eq!( + normalize_http_identity_url("https://youtube.com/watch?v=Z&a=1&b=2").as_deref(), + Some("https://www.youtube.com/watch?a=1&b=2&v=Z") + ); + assert_eq!( + normalize_http_identity_url("https://youtube.com/watch?b=2&a=1&v=Z").as_deref(), + Some("https://www.youtube.com/watch?a=1&b=2&v=Z") + ); + } + + #[test] + fn youtube_embed_to_watch() { + assert_eq!( + normalize_http_identity_url("https://www.youtube.com/embed/dQw4w9WgXcQ").as_deref(), + Some("https://www.youtube.com/watch?v=dQw4w9WgXcQ") + ); + } + + #[test] + fn youtube_shorts_host() { + assert_eq!( + normalize_http_identity_url("https://youtube.com/shorts/AbCdEfGhIjK").as_deref(), + Some("https://www.youtube.com/shorts/AbCdEfGhIjK") + ); + } + + #[test] + fn arbitrary_query_sorted() { + assert_eq!( + normalize_http_identity_url("https://example.com/x?z=1&a=2").as_deref(), + Some("https://example.com/x?a=2&z=1") + ); + } +} Side B — contributor: tommy-mor Side B — commit message: [9e20d06c] Add sorterc dev tool for offline DSL compile and JSONL lint. Introduce a workspace-only binary that validates .sorter files into ranking JSON and scans events.jsonl for corrupt or unreplayable ingests. Co-authored-by: Cursor Side B — unified diff (full patch): diff --git a/Cargo.lock b/Cargo.lock index bf8153d9c723af97122c9ffdd4a7cfe82e853bb6..a07734f089b466440c3ae6fc1087ce85fc24ce62 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1826,6 +1826,17 @@ dependencies = [ "windows-sys 0.60.2", ] +[[package]] +name = "sorterc" +version = "0.0.1" +dependencies = [ + "anyhow", + "clap", + "serde", + "serde_json", + "slugsocial-server", +] + [[package]] name = "spin" version = "0.9.8" diff --git a/Cargo.toml b/Cargo.toml index 149cbf07901eab57c593184ff8719a75d530f1da..25337acdd61e44b20f354c78fed4a88caf896280 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,5 +1,5 @@ [workspace] -members = ["server", "cli"] +members = ["server", "cli", "sorterc"] resolver = "2" diff --git a/agents.md b/agents.md index d8b801e454fdf37e7ac6038b91a69f83b0746d59..ce646ed3cd7123be4732dccec4a6800467e651e7 100644 --- a/agents.md +++ b/agents.md @@ -93,6 +93,17 @@ SLUG_GOOGLE_CLIENT_SECRET=mock After OAuth completes, the pending-session poll returns a `slug_…` bearer token for API calls. +### Dev-only offline tooling + +**`sorterc`** — workspace binary, not published via npm. Compiles `.sorter` files and lints `events.jsonl` without a server: + +``` +cargo run -p sorterc -- compile path/to/doc.sorter [--base events.jsonl] [--room public] [--pretty] +cargo run -p sorterc -- scan path/to/events.jsonl [--pretty] +``` + +`compile` validates DSL, simulates ingest against empty (or `--base`) reducer state, and prints JSON rankings. `scan` reports corrupt JSONL lines and ingests that fail DSL replay. + ### Testing - **Rust tests:** `cargo nextest run --workspace` (163 tests; requires `cargo-nextest`) diff --git a/server/src/lib.rs b/server/src/lib.rs index c1d477d21aea03aff00e6f0689b0b4379d0d68d2..ad8e31099c807fb5844acb16cd5086a2f19327a7 100644 --- a/server/src/lib.rs +++ b/server/src/lib.rs @@ -10,6 +10,7 @@ pub mod form_template; pub mod html; pub mod identity; pub mod middleware; +pub mod offline; pub mod path_types; pub mod ranking; pub mod reducer; diff --git a/server/src/offline.rs b/server/src/offline.rs new file mode 100644 index 0000000000000000000000000000000000000000..54ad0ded096a305ef8454ab2cdd1c3af71b14f5d --- /dev/null +++ b/server/src/offline.rs @@ -0,0 +1,333 @@ +//! Offline `.sorter` compilation and JSONL diagnostics (no network, no auth). + +use std::collections::HashSet; +use std::path::Path; + +use serde::Serialize; +use slug_types::{CheckScopeRanking, RankComponent, RankRow, paths::GardenItemUrl}; + +use crate::{ + api::{resolve_item, validate_ingest_document}, + dsl, + events::{Event, Ingest}, + path_types::ItemId, + reducer::{ReducerState, ScopeId, scope_from_room_wire}, + scope_rank::build_children_rankings, +}; + +#[derive(Debug, Clone, Serialize)] +pub struct CompileStats { + pub items: usize, + pub votes: usize, + pub prose_blocks: usize, +} + +#[derive(Debug, Serialize)] +pub struct CompileResult { + pub ok: bool, + pub threads: Vec, + pub rankings: Vec, + pub stats: CompileStats, +} + +#[derive(Debug, Clone, Serialize)] +pub struct CompileError { + pub ok: bool, + pub error: String, + #[serde(skip_serializing_if = "Option::is_none")] + pub hint: Option, +} + +#[derive(Debug, Clone, Serialize)] +pub struct BadJsonLine { + pub line: usize, + pub message: String, +} + +#[derive(Debug, Clone, Serialize)] +pub struct MalformedIngest { + pub line: usize, + pub id: String, + pub room_id: String, + pub thread_tag: String, + pub reason: String, +} + +#[derive(Debug, Clone, Serialize)] +pub struct ScanResult { + pub ok: bool, + pub path: String, + pub total_lines: usize, + pub parsed_events: usize, + pub bad_json_lines: Vec, + pub malformed_ingests: Vec, + pub skipped_ingests: usize, +} + +fn document_stats(doc: &dsl::Document) -> CompileStats { + let mut items = 0usize; + let mut votes = 0usize; + let mut prose_blocks = 0usize; + for stmt in &doc.statements { + match stmt { + dsl::Stmt::Item { .. } => items += 1, + dsl::Stmt::Vote { .. } => votes += 1, + dsl::Stmt::Prose { .. } => prose_blocks += 1, + } + } + CompileStats { + items, + votes, + prose_blocks, + } +} + +fn threads_in_document(text: &str) -> Vec { + let mut out = HashSet::new(); + for line in text.lines() { + let trimmed = line.trim(); + if !trimmed.starts_with('#') { + continue; + } + let rest = trimmed.trim_start_matches('#').trim(); + if rest.is_empty() { + continue; + } + let tag = rest.split_whitespace().next().unwrap_or(rest); + let tag = tag.split(':').next().unwrap_or(tag).trim(); + if tag.is_empty() { + continue; + } + out.insert(format!("#{}", crate::canonical_path::canonicalize_tag(tag))); + } + let mut tags: Vec = out.into_iter().collect(); + tags.sort(); + tags +} + +fn voted_parent_scopes(doc: &dsl::Document) -> Vec { + let mut parents = HashSet::new(); + for stmt in &doc.statements { + if let dsl::Stmt::Vote { item1, item2, .. } = stmt { + if let (Ok(a), Ok(b)) = (resolve_item(item1), resolve_item(item2)) { + if let Some(p) = a.parent() { + parents.insert(p); + } + if let Some(p) = b.parent() { + parents.insert(p); + } + } + } + } + let mut out: Vec = parents.into_iter().collect(); + out.sort(); + out +} + +fn rankings_for_simulated( + simulated: &ReducerState, + scope: &ScopeId, + room_wire: &str, + doc: &dsl::Document, +) -> Vec { + voted_parent_scopes(doc) + .iter() + .map(|parent| { + let scoped_content = simulated + .content_for_scope(&scope) + .unwrap_or_else(|| simulated.public()); + let scoped = build_children_rankings(scoped_content, parent); + let components: Vec = scoped + .component_rankings + .into_iter() + .map(|comp| RankComponent { + pairs: comp.pairs, + ranking: comp + .ranked + .into_iter() + .map(|r| RankRow { + item: GardenItemUrl::from_stored(&r.item, room_wire), + score: r.score, + percent: None, + }) + .collect(), + }) + .collect(); + CheckScopeRanking { + parent: GardenItemUrl::from_stored(parent, room_wire).into_inner(), + components, + unranked_items: scoped + .unranked_items + .into_iter() + .map(|it| GardenItemUrl::from_stored(&it, room_wire)) + .collect(), + } + }) + .collect() +} + +/// Validate and simulate one `.sorter` document against optional base reducer state. +pub fn compile_document( + base: &ReducerState, + room: &str, + text: &str, +) -> Result { + let room_key = room.trim(); + let scope = scope_from_room_wire(room_key); + let validated = validate_ingest_document(base, text, &scope).map_err(|(_, message, hint)| { + CompileError { + ok: false, + error: message, + hint, + } + })?; + + let event = Event::Ingest(Ingest { + ts: validated.ts, + id: uuid::Uuid::new_v4().to_string(), + raw: validated.raw_text.clone(), + principal: "offline".to_string(), + delegate: None, + room_id: room_key.to_string(), + thread_tag: "offline".to_string(), + }); + + let mut simulated = base.clone(); + simulated.apply_event(event); + + Ok(CompileResult { + ok: true, + threads: threads_in_document(text), + rankings: rankings_for_simulated(&simulated, &scope, room_key, &validated.doc), + stats: document_stats(&validated.doc), + }) +} + +fn ingest_parse_error(raw: &str) -> Option { + dsl::parse_full(raw).err().map(|e| e.to_string()) +} + +fn load_events_from_jsonl(path: &Path) -> Result<(Vec<(usize, Event)>, Vec), std::io::Error> { + let text = std::fs::read_to_string(path)?; + let mut events = Vec::new(); + let mut bad_json_lines = Vec::new(); + for (idx, line) in text.lines().enumerate() { + let line_no = idx + 1; + let trimmed = line.trim(); + if trimmed.is_empty() { + continue; + } + match serde_json::from_str::(trimmed) { + Ok(ev) => events.push((line_no, ev)), + Err(e) => bad_json_lines.push(BadJsonLine { + line: line_no, + message: e.to_string(), + }), + } + } + Ok((events, bad_json_lines)) +} + +/// Replay a JSONL event log into reducer state (same rules as server boot). +pub fn load_reducer_from_jsonl(path: &Path) -> Result<(ReducerState, Vec), std::io::Error> { + let (events, bad_json_lines) = load_events_from_jsonl(path)?; + let mut state = ReducerState::default(); + for (_line_no, ev) in events { + state.apply_event(ev); + } + Ok((state, bad_json_lines)) +} + +/// Scan an events.jsonl for corrupt JSON lines and ingests that fail DSL replay. +pub fn scan_jsonl(path: &Path) -> Result { + let text = std::fs::read_to_string(path)?; + let total_lines = text.lines().count(); + let (events, bad_json_lines) = load_events_from_jsonl(path)?; + + let mut malformed_ingests = Vec::new(); + let mut skipped_ingests = 0usize; + let mut state = ReducerState::default(); + let parsed_events = events.len(); + + for (line_no, ev) in events { + if let Event::Ingest(ref ing) = ev { + if let Some(reason) = ingest_parse_error(&ing.raw) { + malformed_ingests.push(MalformedIngest { + line: line_no, + id: ing.id.clone(), + room_id: ing.room_id.clone(), + thread_tag: ing.thread_tag.clone(), + reason, + }); + } + let before = state.ingests_by_id.len(); + state.apply_event(ev); + if state.ingests_by_id.len() == before { + skipped_ingests += 1; + } + } else { + state.apply_event(ev); + } + } + + let ok = bad_json_lines.is_empty() && malformed_ingests.is_empty() && skipped_ingests == 0; + + Ok(ScanResult { + ok, + path: path.display().to_string(), + total_lines, + parsed_events, + bad_json_lines, + malformed_ingests, + skipped_ingests, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + const TUTORIAL: &str = include_str!("../tests/fixtures/tutorial.sorter"); + + #[test] + fn compile_tutorial_fixture_emits_rankings() { + let result = compile_document(&ReducerState::default(), "public", TUTORIAL).unwrap(); + assert!(result.ok); + assert!(!result.threads.is_empty()); + assert!(result.stats.items >= 6); + assert!(result.stats.votes >= 6); + assert!(!result.rankings.is_empty()); + } + + #[test] + fn compile_rejects_vote_on_missing_item() { + let err = compile_document( + &ReducerState::default(), + "public", + "{ reason }\n~/missing/a 2:1 ~/missing/b", + ) + .unwrap_err(); + assert!(!err.ok); + assert!(err.error.contains("undefined")); + } + + #[test] + fn scan_empty_jsonl_is_ok() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("events.jsonl"); + std::fs::write(&path, "").unwrap(); + let report = scan_jsonl(&path).unwrap(); + assert!(report.ok); + assert!(report.bad_json_lines.is_empty()); + } + + #[test] + fn scan_reports_bad_json_line() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("events.jsonl"); + std::fs::write(&path, "{not json}\n").unwrap(); + let report = scan_jsonl(&path).unwrap(); + assert!(!report.ok); + assert_eq!(report.bad_json_lines.len(), 1); + } +} diff --git a/sorterc/Cargo.toml b/sorterc/Cargo.toml new file mode 100644 index 0000000000000000000000000000000000000000..92d477aff9b53291fb1a266db83065c1c800791d --- /dev/null +++ b/sorterc/Cargo.toml @@ -0,0 +1,18 @@ +[package] +name = "sorterc" +version = "0.0.1" +edition = "2021" +license = "MIT" +publish = false +description = "Offline .sorter compiler and events.jsonl linter (dev only)" + +[[bin]] +name = "sorterc" +path = "src/main.rs" + +[dependencies] +anyhow = "1" +clap = { version = "4", features = ["derive"] } +serde = { version = "1", features = ["derive"] } +serde_json = "1" +slugsocial-server = { path = "../server" } diff --git a/sorterc/readme.md b/sorterc/readme.md new file mode 100644 index 0000000000000000000000000000000000000000..1ebcc3fc935541ea9e47e0458ec67a750fe19fff --- /dev/null +++ b/sorterc/readme.md @@ -0,0 +1,92 @@ +# sorterc + +Dev-only offline tooling for the slug `.sorter` DSL and `events.jsonl` event log. + +`sorterc` is **not** published via npm and does not talk to slug.social. It reuses the same parser, validator, and ranking code as the server, but runs entirely on local files. + +## Build + +From the repo root: + +```bash +cargo build -p sorterc +cargo run -p sorterc -- --help +``` + +## Commands + +### `compile` — evaluate a `.sorter` document + +Reads a `.sorter` file (or `-` for stdin), validates the DSL, simulates one ingest against reducer state, and prints JSON rankings to stdout. + +```bash +cargo run -p sorterc -- compile path/to/doc.sorter +cargo run -p sorterc -- compile path/to/doc.sorter --pretty +cargo run -p sorterc -- compile - --pretty # stdin +cargo run -p sorterc -- compile doc.sorter --base events.jsonl # seed garden from log +cargo run -p sorterc -- compile doc.sorter --room public # default room +``` + +**Flags** + +| Flag | Description | +|------|-------------| +| `--base PATH` | Replay an `events.jsonl` first, then compile against that garden state | +| `--room ID` | Room wire id (`public` or private room id). Default: `public` | +| `--pretty` | Pretty-print JSON | + +**Success output** (shape): + +```json +{ + "ok": true, + "threads": ["#my-thread"], + "rankings": [ … ], + "stats": { "items": 3, "votes": 2, "prose_blocks": 5 } +} +``` + +Rankings use the same structure as the server's dry-run check: parent scope, connected components, scores, unranked items. + +**Error output** exits with code 1: + +```json +{ + "ok": false, + "error": "parse error", + "hint": "…" +} +``` + +### `scan` — lint an `events.jsonl` + +Reads a JSONL event log and reports problems without starting a server. + +```bash +cargo run -p sorterc -- scan events.jsonl +cargo run -p sorterc -- scan events.jsonl --pretty +``` + +Reports: + +- **bad JSON lines** — lines that are not valid JSON +- **malformed ingests** — ingest events whose `raw` DSL fails to parse +- **skipped ingests** — ingests dropped during replay (same behavior as server boot) + +Exits 0 when clean, 1 when any issue is found. + +## Typical uses + +- Iterate on `.sorter` files in an editor and pipe through `compile` to see rankings instantly +- Verify a downloaded or edited `events.jsonl` before uploading to Fly +- Debug "malformed ingest" warnings from production boot logs +- CI or pre-commit checks on fixture docs (no OAuth, no network) + +## What it does not do + +- Post to slug.social or append to a live log +- Authenticate users or bind agents +- Run browser/UI tests +- Replace `slugsocial public check` for operators who want the full RPC path against a running server + +For live server dry-run against current garden state, use `npx slugsocial public check` or `POST /try/check` in the browser. diff --git a/sorterc/src/main.rs b/sorterc/src/main.rs new file mode 100644 index 0000000000000000000000000000000000000000..71382c180085cb0ad71043c852f8db5d3a48a284 --- /dev/null +++ b/sorterc/src/main.rs @@ -0,0 +1,117 @@ +use std::path::{Path, PathBuf}; + +use anyhow::{bail, Context, Result}; +use clap::{Parser, Subcommand}; +use slugsocial_server::{ + offline::{self, CompileError, CompileResult, ScanResult}, + reducer::ReducerState, +}; + +#[derive(Parser)] +#[command( + name = "sorterc", + about = "Offline .sorter compiler and events.jsonl linter (dev only)", + version +)] +struct Cli { + #[command(subcommand)] + cmd: Command, +} + +#[derive(Subcommand)] +enum Command { + /// Parse and simulate a .sorter document; emit ranking JSON to stdout. + Compile { + /// `.sorter` file, or `-` for stdin. + file: PathBuf, + /// Room wire id (`public` or private room id). + #[arg(long, default_value = "public")] + room: String, + /// Optional events.jsonl to replay before compiling (seed garden state). + #[arg(long)] + base: Option, + /// Pretty-print JSON. + #[arg(long)] + pretty: bool, + }, + /// Scan an events.jsonl for corrupt JSON lines and malformed ingests. + Scan { + file: PathBuf, + #[arg(long)] + pretty: bool, + }, +} + +fn read_input(path: &Path) -> Result { + if path.as_os_str() == "-" { + use std::io::Read; + let mut buf = String::new(); + std::io::stdin().read_to_string(&mut buf)?; + Ok(buf) + } else { + std::fs::read_to_string(path) + .with_context(|| format!("read {}", path.display())) + } +} + +fn load_base_state(base: Option<&Path>) -> Result { + let Some(path) = base else { + return Ok(ReducerState::default()); + }; + let (state, bad_lines) = offline::load_reducer_from_jsonl(path) + .with_context(|| format!("load base jsonl {}", path.display()))?; + if !bad_lines.is_empty() { + bail!( + "base jsonl has {} corrupt line(s); fix or omit --base", + bad_lines.len() + ); + } + Ok(state) +} + +fn print_json(value: &T, pretty: bool) -> Result<()> { + if pretty { + println!("{}", serde_json::to_string_pretty(value)?); + } else { + println!("{}", serde_json::to_string(value)?); + } + Ok(()) +} + +fn run_compile(file: PathBuf, room: String, base: Option, pretty: bool) -> Result<()> { + let text = read_input(&file)?; + let base_state = load_base_state(base.as_deref())?; + match offline::compile_document(&base_state, &room, &text) { + Ok(result) => { + print_json::(&result, pretty)?; + Ok(()) + } + Err(err) => { + print_json::(&err, pretty)?; + std::process::exit(1); + } + } +} + +fn run_scan(file: PathBuf, pretty: bool) -> Result<()> { + let report = offline::scan_jsonl(&file) + .with_context(|| format!("scan {}", file.display()))?; + print_json::(&report, pretty)?; + if !report.ok { + std::process::exit(1); + } + Ok(()) +} + +fn main() -> Result<()> { + let cli = Cli::parse(); + match cli.cmd { + Command::Compile { + file, + room, + base, + pretty, + } => run_compile(file, room, base, pretty), + Command::Scan { file, pretty } => run_scan(file, pretty), + } +}