You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [a888d56c] refactor: centralize path identity in slug-types Move canonicalization and CanonicalItemUrl into types::paths with GardenItemUrl, ForumThreadUrl, and TildeOntologyPath for JSON hrefs. Server canonical_path and path_types re-export slug-types; RPC and validation build hrefs via those types instead of string helpers. Made-with: Cursor Side A — unified diff (full patch): diff --git a/server/src/api/helpers.rs b/server/src/api/helpers.rs index 9b71491e9f9efc44a2a4beba09be8f64bd2ff2ee..03b3e77911ccd662bec8635345dafe2593cf242e 100644 --- a/server/src/api/helpers.rs +++ b/server/src/api/helpers.rs @@ -4,12 +4,12 @@ use axum::{ Json, }; use sha2::{Digest, Sha256}; +use slug_types::paths::{CanonicalItemUrl, GardenItemUrl}; use slug_types::*; use std::collections::HashMap; use crate::{ canonical_path::canonicalize_item, - path_types::CanonicalItemUrl, ranking::connected_components_from_voted_pairs, }; @@ -30,64 +30,6 @@ pub fn now_ms() -> i64 { t.as_millis() as i64 } -/// Serialize a canonical item for JSON: absolute URLs stay as-is; bare paths get a `/` prefix. -pub fn item_path_for_api(item: &str) -> String { - if item.starts_with("http://") || item.starts_with("https://") { - item.to_string() - } else { - format!("/{}", item) - } -} - -/// Same as [`item_path_for_api`], but for private rooms ontology items are prefixed with -/// `/r/{short}/{slug}` so the URL matches the web app (`/r/…/~/…` routes). -pub fn item_path_for_api_in_room(item: &str, room_wire: &str) -> String { - let room = room_wire.trim(); - if room.is_empty() || room == "public" { - return item_path_for_api(item); - } - let Some((short, slug)) = room.split_once('/') else { - return item_path_for_api(item); - }; - if short.is_empty() || slug.is_empty() { - return item_path_for_api(item); - } - let Some(c) = CanonicalItemUrl::parse(item) else { - return item_path_for_api(item); - }; - let root = CanonicalItemUrl::ontology_root(); - let item_norm = c.as_str().trim_end_matches('/'); - let root_norm = root.as_str().trim_end_matches('/'); - if let Some(tail) = c.tilde_tail() { - return if tail.is_empty() { - format!("https://slug.social/r/{short}/{slug}/~") - } else { - format!("https://slug.social/r/{short}/{slug}/~/{}", tail) - }; - } - if item_norm == root_norm { - return format!("https://slug.social/r/{short}/{slug}/~"); - } - item_path_for_api(item) -} - -/// Absolute thread URL for forum JSON (`/t/…` vs `/r/…/t/…`). -pub fn forum_thread_web_url(room_wire: &str, thread_tag: &str) -> String { - let room = room_wire.trim(); - let tag = thread_tag.trim().trim_start_matches('#'); - if room.is_empty() || room == "public" { - format!("https://slug.social/t/{tag}") - } else if let Some((short, slug)) = room.split_once('/') { - if short.is_empty() || slug.is_empty() { - format!("https://slug.social/t/{tag}") - } else { - format!("https://slug.social/r/{short}/{slug}/t/{tag}") - } - } else { - format!("https://slug.social/t/{tag}") - } -} - /// Resolve an item path as a first-class canonical path. pub fn resolve_item(item: &str) -> Result { let canonical = canonicalize_item(item); @@ -109,14 +51,12 @@ pub fn parse_parent_specs(parent: Option<&String>) -> Vec { } /// Apply offset+limit pagination to the flattened component rankings. -/// Items are flattened in component order (largest component first), then unranked last. -/// Returns (components, unranked_items) after the window. pub fn paginate_rankings( components: Vec, - unranked_items: Vec, + unranked_items: Vec, offset: usize, limit: Option, -) -> (Vec, Vec) { +) -> (Vec, Vec) { let mut remaining_skip = offset; let mut remaining_take = limit.unwrap_or(usize::MAX); let mut out_components: Vec = Vec::new(); @@ -141,7 +81,7 @@ pub fn paginate_rankings( }); } - let out_unranked: Vec = if remaining_take > 0 { + let out_unranked: Vec = if remaining_take > 0 { unranked_items .into_iter() .skip(remaining_skip) @@ -183,11 +123,9 @@ pub fn is_pair_voted(group: &crate::reducer::GroupState, a: &str, b: &str) -> bo group.voted_pairs.contains(&(i, j)) } -/// Compute graph connectivity stats for a set of items within the ranking group. pub fn compute_connectivity_stats(group: &crate::reducer::GroupState, pool: &[String]) -> ConnectivityStats { let n = pool.len(); - // Map pool items to global indices (items not yet in the group get no index) let global_idxs: Vec> = pool .iter() .map(|it| { @@ -197,7 +135,6 @@ pub fn compute_connectivity_stats(group: &crate::reducer::GroupState, pool: &[St .collect(); let present: Vec = global_idxs.iter().filter_map(|x| *x).collect(); - // Build local index mapping for items that exist in the ranking group let global_to_local: HashMap = present .iter() .enumerate() @@ -213,7 +150,6 @@ pub fn compute_connectivity_stats(group: &crate::reducer::GroupState, pool: &[St }), ); - // Items not in the ranking group at all are also isolates let items_not_in_group = global_idxs.iter().filter(|x| x.is_none()).count(); let num_components = comps.len() + isolates.len() + items_not_in_group; @@ -237,52 +173,3 @@ pub fn vote_touches_path(a: &str, b: &str, parent_canon: &str) -> bool { let under = |item: &str| item == parent_canon || item.starts_with(&format!("{}/", parent_canon)); under(a) || under(b) } - -#[cfg(test)] -mod wire_url_tests { - use super::{forum_thread_web_url, item_path_for_api_in_room}; - - #[test] - fn public_room_unchanged() { - let u = "https://slug.social/~/a/b"; - assert_eq!(item_path_for_api_in_room(u, "public"), u); - } - - #[test] - fn private_room_prefixes_ontology() { - assert_eq!( - item_path_for_api_in_room("https://slug.social/~/topic/x", "9ab12cd/my-room"), - "https://slug.social/r/9ab12cd/my-room/~/topic/x" - ); - } - - #[test] - fn private_room_ontology_root() { - assert_eq!( - item_path_for_api_in_room("https://slug.social/~", "9ab12cd/my-room"), - "https://slug.social/r/9ab12cd/my-room/~" - ); - assert_eq!( - item_path_for_api_in_room("https://slug.social/~/", "9ab12cd/my-room"), - "https://slug.social/r/9ab12cd/my-room/~" - ); - } - - #[test] - fn external_url_untouched_in_private_room() { - let u = "https://example.com/z"; - assert_eq!(item_path_for_api_in_room(u, "9ab12cd/my-room"), u); - } - - #[test] - fn forum_web_public_vs_room() { - assert_eq!( - forum_thread_web_url("public", "debate"), - "https://slug.social/t/debate" - ); - assert_eq!( - forum_thread_web_url("9ab12cd/my-room", "#debate"), - "https://slug.social/r/9ab12cd/my-room/t/debate" - ); - } -} diff --git a/server/src/api/mod.rs b/server/src/api/mod.rs index 042aa248305f9362a3be78f9eea2a5abf6ba707a..cf22cb0129366c3aed031bc86f3197a4321cb806 100644 --- a/server/src/api/mod.rs +++ b/server/src/api/mod.rs @@ -24,8 +24,7 @@ pub use auth::{ pub use helpers::{ api_error, compute_connectivity_stats, is_pair_voted, now_ms, paginate_rankings, - parse_parent_specs, pick_random_distinct, sha256_hex, resolve_item, vote_touches_path, - item_path_for_api, + parse_parent_specs, pick_random_distinct, resolve_item, sha256_hex, vote_touches_path, }; pub use rpc::handle_rpc_batch; diff --git a/server/src/api/rpc.rs b/server/src/api/rpc.rs index 5b91f5836625eedbb1cd9423168046e3fb576c17..5f7d50188f1381267402f2e57e671234ef5db2fd 100644 --- a/server/src/api/rpc.rs +++ b/server/src/api/rpc.rs @@ -8,6 +8,7 @@ use axum::{ Json, }; use rand::seq::SliceRandom; +use slug_types::paths::{ForumThreadUrl, GardenItemUrl, TildeOntologyPath}; use slug_types::*; use crate::{ @@ -27,9 +28,8 @@ use crate::{ use super::auth::verify_bearer_principal; use super::helpers::{ - compute_connectivity_stats, forum_thread_web_url, is_pair_voted, item_path_for_api, - item_path_for_api_in_room, now_ms, paginate_rankings, parse_parent_specs, pick_random_distinct, - resolve_item, vote_touches_path, + compute_connectivity_stats, is_pair_voted, now_ms, paginate_rankings, parse_parent_specs, + pick_random_distinct, resolve_item, vote_touches_path, }; use super::validate::{normalize_room_and_thread, validate_ingest_document}; @@ -184,7 +184,7 @@ fn compute_scope_rank_changes( }; if changed { changes.push(RankChange { - item: item_path_for_api_in_room(&item, room_wire), + item: GardenItemUrl::from_storage_str(&item, room_wire), before: b, after: a, }); @@ -206,7 +206,7 @@ fn compute_scope_rank_changes( parent: if parent.is_empty() { "/".to_string() } else { - item_path_for_api_in_room(parent, room_wire) + GardenItemUrl::from_storage_str(parent, room_wire).into_inner() }, changes, }) @@ -302,7 +302,7 @@ fn build_rank_response_for_content( .ranked .into_iter() .map(|r| RankRow { - item: item_path_for_api_in_room(r.item.as_str(), room_wire), + item: GardenItemUrl::from_stored(&r.item, room_wire), percent: if want_percent { Some((r.score / max_score) * 100.0) } else { @@ -315,10 +315,10 @@ fn build_rank_response_for_content( }) .collect(); - let prefixed_unranked: Vec = rankings + let prefixed_unranked: Vec = rankings .unranked_items .into_iter() - .map(|s| item_path_for_api_in_room(s.as_str(), room_wire)) + .map(|s| GardenItemUrl::from_stored(&s, room_wire)) .collect(); let (components, unranked_items) = if offset > 0 || limit.is_some() { @@ -537,13 +537,13 @@ async fn rpc_post( ( "npx slugsocial public garden pair".to_string(), "npx slugsocial public garden rank".to_string(), - forum_thread_web_url("public", &thread_id), + ForumThreadUrl::from_room_tag("public", &thread_id), ) } else { ( format!("npx slugsocial private {room_key} garden pair"), format!("npx slugsocial private {room_key} garden rank"), - forum_thread_web_url(&room_key, &thread_id), + ForumThreadUrl::from_room_tag(&room_key, &thread_id), ) }; @@ -664,7 +664,7 @@ async fn rpc_check( .ranked .into_iter() .map(|r| RankRow { - item: item_path_for_api_in_room(r.item.as_str(), &room_key), + item: GardenItemUrl::from_stored(&r.item, &room_key), score: r.score, percent: None, }) @@ -672,12 +672,12 @@ async fn rpc_check( }) .collect(); CheckScopeRanking { - parent: item_path_for_api_in_room(parent.as_str(), &room_key), + parent: GardenItemUrl::from_stored(parent, &room_key).into_inner(), components, unranked_items: scoped .unranked_items .into_iter() - .map(|it| item_path_for_api_in_room(it.as_str(), &room_key)) + .map(|it| GardenItemUrl::from_stored(&it, &room_key)) .collect(), } }) @@ -687,13 +687,13 @@ async fn rpc_check( vec![ "npx slugsocial public forum post --delegate ".to_string(), "npx slugsocial public forum list".to_string(), - forum_thread_web_url("public", &thread_id), + ForumThreadUrl::from_room_tag("public", &thread_id).into_inner(), ] } else { vec![ format!("npx slugsocial private {room_key} forum post --delegate "), format!("npx slugsocial private {room_key} forum list"), - forum_thread_web_url(&room_key, &thread_id), + ForumThreadUrl::from_room_tag(&room_key, &thread_id).into_inner(), ] }; @@ -717,7 +717,7 @@ fn rpc_list_forum_threads(reduced: &ReducerState, room: &str) -> ThreadsResponse .map(|((_, tag), ts)| ThreadSummary { thread: format!("#{tag}"), last_activity_ts: ts.last_activity_ts, - web: forum_thread_web_url(room, tag), + web: ForumThreadUrl::from_room_tag(room, tag), }) .collect(); out.sort_by(|a, b| b.last_activity_ts.cmp(&a.last_activity_ts)); @@ -885,7 +885,7 @@ fn rpc_search(reduced: &ReducerState, q: &str, limit: usize, principal: Option<& } if score > 0 { scored_items.push((score, SearchItemHit { - path: item_path_for_api(item.as_str()), + path: GardenItemUrl::from_storage_str(item.as_str(), "public"), body: content.item_bodies.get(item).map(|b| snippet_around(b, &words, 120)), })); } @@ -1058,8 +1058,8 @@ async fn rpc_get_pair(state: &AppState, room: String, parent_path: String) -> Re .collect(); let cs = compute_connectivity_stats(&content.ranking_group, &pool); Ok(RpcResult::Pair(PairResponse { - left: item_path_for_api_in_room(&left, &room), - right: item_path_for_api_in_room(&right, &room), + left: GardenItemUrl::from_storage_str(&left, &room), + right: GardenItemUrl::from_storage_str(&right, &room), left_body: lb, right_body: rb, threads: th, @@ -1137,7 +1137,7 @@ pub async fn handle_rpc_batch( if !content.items.contains(&item) { line_err( "item not found", - Some(format!("{} does not exist", item_path_for_api_in_room(&item_str, &room))), + Some(format!("{} does not exist", GardenItemUrl::from_storage_str(&item_str, &room))), ) } else { const MAX_ITEM_BODY: usize = 10_000; @@ -1160,7 +1160,7 @@ pub async fn handle_rpc_batch( .map(|s| s.iter().cloned().collect()) .unwrap_or_default(); line_ok(RpcResult::GardenItem(ItemResponse { - item: item_path_for_api_in_room(&item_str, &room), + item: GardenItemUrl::from_storage_str(&item_str, &room), body, truncated, body_len, @@ -1554,7 +1554,7 @@ pub async fn handle_rpc_batch( for r in items { let pct = want_percent.then(|| ((r.score - bot) / range * 100.0).clamp(0.0, 100.0)); ranked.push(RankRow { - item: item_path_for_api_in_room(r.item.as_str(), &room), + item: GardenItemUrl::from_storage_str(r.item.as_str(), &room), score: r.score, percent: pct, }); @@ -1574,7 +1574,7 @@ pub async fn handle_rpc_batch( let page: Vec = ranked .into_iter() .chain(unranked.into_iter().map(|it| RankRow { - item: item_path_for_api_in_room(&it, &room), + item: GardenItemUrl::from_storage_str(&it, &room), score: 0.0, percent: want_percent.then_some(0.0), })) @@ -1618,7 +1618,7 @@ pub async fn handle_rpc_batch( if !content.items.contains(&item) { line_err( "item not found", - Some(format!("{} does not exist", item_path_for_api_in_room(&item_str, &room))), + Some(format!("{} does not exist", GardenItemUrl::from_storage_str(&item_str, &room))), ) } else { let votes: Vec = content @@ -1629,8 +1629,8 @@ pub async fn handle_rpc_batch( .take(limit) .map(|v| VoteRow { ts: v.ts, - a: item_path_for_api_in_room(v.a.as_str(), &room), - b: item_path_for_api_in_room(v.b.as_str(), &room), + a: GardenItemUrl::from_stored(&v.a, &room), + b: GardenItemUrl::from_stored(&v.b, &room), ratio: format!("{}:{}", v.ratio_left, v.ratio_right), actor: Some(v.principal.clone()), body: v.body.clone(), @@ -1640,7 +1640,7 @@ pub async fn handle_rpc_batch( }) .unwrap_or_default(); line_ok(RpcResult::Matchup(MatchupResponse { - item: item_path_for_api_in_room(&item_str, &room), + item: GardenItemUrl::from_storage_str(&item_str, &room), votes, })) } @@ -1667,8 +1667,8 @@ pub async fn handle_rpc_batch( if a == item_str || b == item_str { Some(VoteRow { ts: e.ts, - a: item_path_for_api_in_room(&a, &room), - b: item_path_for_api_in_room(&b, &room), + a: GardenItemUrl::from_storage_str(&a, &room), + b: GardenItemUrl::from_storage_str(&b, &room), ratio: format!("{}:{}", ratio_left, ratio_right), actor: reduced.ingests_by_id.get(&e.post_id).map(|ing| ing.principal.clone()), body: explanation, @@ -1704,7 +1704,7 @@ pub async fn handle_rpc_batch( } }).collect(); line_ok(RpcResult::RankHistory(RankHistoryResponse { - item: item_path_for_api_in_room(&item_str, &room), + item: GardenItemUrl::from_storage_str(&item_str, &room), history, })) } @@ -1716,17 +1716,13 @@ pub async fn handle_rpc_batch( } else { let content = content_for_room(&reduced, &room); let parents: HashSet<&str> = content.item_children.keys().map(|s| s.as_str()).collect(); - let mut paths: Vec = content + let mut paths: Vec = content .items .iter() .filter(|p| !parents.contains(p.as_str())) - .map(|p| p.as_str().to_string()) - .collect(); - paths.sort(); - let paths: Vec = paths - .into_iter() - .map(|p| item_path_for_api_in_room(&p, &room)) + .map(|p| GardenItemUrl::from_stored(p, &room)) .collect(); + paths.sort_by(|a, b| a.as_str().cmp(b.as_str())); line_ok(RpcResult::Leaves(LeavesResponse { paths })) } }, @@ -1743,24 +1739,13 @@ pub async fn handle_rpc_batch( let mut v: Vec = roots.iter() .map(|path| { let children = content.item_children.get(path.as_str()).map(|s| s.len()).unwrap_or(0); - let path_label = CanonicalItemUrl::parse(path.as_str()) - .and_then(|c| { - c.tilde_tail().map(|t| { - if t.is_empty() { - "~/".to_string() - } else { - format!("~/{}", t) - } - }) - }) - .unwrap_or_else(|| path.to_string()); PathSummary { - path: path_label, + path: TildeOntologyPath::from_stored(path), children, - web: item_path_for_api_in_room(path.as_str(), &room), + web: GardenItemUrl::from_stored(path, &room), } }).collect(); - v.sort_by(|a, b| a.path.cmp(&b.path)); + v.sort_by(|a, b| a.path.as_str().cmp(b.path.as_str())); v }) .unwrap_or_default(); @@ -1790,8 +1775,8 @@ pub async fn handle_rpc_batch( .take(limit) .map(|v| VoteRow { ts: v.ts, - a: item_path_for_api_in_room(v.a.as_str(), &room), - b: item_path_for_api_in_room(v.b.as_str(), &room), + a: GardenItemUrl::from_stored(&v.a, &room), + b: GardenItemUrl::from_stored(&v.b, &room), ratio: format!("{}:{}", v.ratio_left, v.ratio_right), actor: Some(v.principal.clone()), body: v.body.clone(), diff --git a/server/src/api/validate.rs b/server/src/api/validate.rs index 27150577982f52cbd6d11bb93654dc3d4cf75cc5..a51c783ee9785569b5a44c0b1572471fe00d174b 100644 --- a/server/src/api/validate.rs +++ b/server/src/api/validate.rs @@ -7,8 +7,9 @@ use crate::{ path_types::CanonicalItemUrl, reducer::{ReducerState, ScopeId}, }; +use slug_types::paths::GardenItemUrl; -use super::helpers::{item_path_for_api, resolve_item}; +use super::helpers::resolve_item; #[derive(Debug)] pub struct ValidatedIngest { @@ -22,6 +23,10 @@ pub fn validate_ingest_document( text: &str, scope: &ScopeId, ) -> Result)> { + let room_wire = match scope { + ScopeId::Public => "public", + ScopeId::Room(r) => r.as_str(), + }; let public_content = reduced.public(); let scoped_content = match scope { ScopeId::Public => None, @@ -61,14 +66,14 @@ pub fn validate_ingest_document( let Some(body_text) = body else { return Err(( StatusCode::BAD_REQUEST, - format!("item missing body: {}", item_path_for_api(&item)), + format!("item missing body: {}", GardenItemUrl::from_storage_str(&item, room_wire)), Some("items must be declared with bodies, e.g. `~/path/item { ... }`".to_string()), )); }; if body_text.trim().is_empty() { return Err(( StatusCode::BAD_REQUEST, - format!("item body is empty: {}", item_path_for_api(&item)), + format!("item body is empty: {}", GardenItemUrl::from_storage_str(&item, room_wire)), Some("write at least one sentence inside `{ ... }`".to_string()), )); } @@ -101,7 +106,7 @@ pub fn validate_ingest_document( let key = CanonicalItemUrl((*it).clone()); !defined_in_doc.contains(*it) && !item_exists(&key) }) - .map(|it| item_path_for_api(it)) + .map(|it| GardenItemUrl::from_storage_str(it, room_wire).into_inner()) .collect(); if !missing.is_empty() { return Err(( @@ -119,7 +124,7 @@ pub fn validate_ingest_document( let key = CanonicalItemUrl((*it).clone()); !defined_in_doc.contains(*it) && !body_exists(&key) }) - .map(|it| item_path_for_api(it)) + .map(|it| GardenItemUrl::from_storage_str(it, room_wire).into_inner()) .collect(); if !missing_body.is_empty() { return Err(( diff --git a/server/src/canonical_path.rs b/server/src/canonical_path.rs index 5c0febe883d8b3978a25896ed0d71df8509a8717..8a1998025121838b0867f8c5ddbd28aea41a23d2 100644 --- a/server/src/canonical_path.rs +++ b/server/src/canonical_path.rs @@ -1,92 +1,3 @@ -//! Normalization for thread tags and ontology item URLs (DSL ↔ stored canonical form). -//! Not event types — see `events` and `path_types`. +//! Re-exports — implementations live in `slug-types` (`paths` module). -/// Thread / public tag: stored without leading `#`, lowercase. -pub fn canonicalize_tag(input: &str) -> String { - input.trim().trim_start_matches('#').to_lowercase() -} - -/// Ontology item reference → canonical absolute URL on the slug host. -pub fn canonicalize_item(input: &str) -> String { - let s = input.trim(); - if s.is_empty() { - return String::new(); - } - - if let Some(rest) = s.strip_prefix("https://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let host = host.trim().to_lowercase(); - if tail.is_empty() { - return format!("https://{}", host); - } else { - return format!("https://{}/{}", host, tail); - } - } - if let Some(rest) = s.strip_prefix("http://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let host = host.trim().to_lowercase(); - if tail.is_empty() { - return format!("http://{}", host); - } else { - return format!("http://{}/{}", host, tail); - } - } - - let is_tilde = s.starts_with("~/"); - let rest = s.strip_prefix("~/").or_else(|| s.strip_prefix("/")).unwrap_or(s); - - let tail = rest - .split('/') - .filter_map(|seg| { - let t = seg.trim(); - if t.is_empty() { - None - } else { - Some(t.to_lowercase()) - } - }) - .collect::>() - .join("/"); - - if is_tilde { - format!("https://slug.social/~/{}", tail) - } else if tail.is_empty() { - "https://slug.social".to_string() - } else { - format!("https://slug.social/{}", tail) - } -} - -pub fn item_path_segments(input: &str) -> Vec { - let canonical = canonicalize_item(input); - if canonical.is_empty() { - return vec![]; - } - - if let Some(rest) = canonical.strip_prefix("https://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let mut out = vec![format!("https://{}", host)]; - out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); - return out; - } - if let Some(rest) = canonical.strip_prefix("http://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let mut out = vec![format!("http://{}", host)]; - out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); - return out; - } - - canonical - .split('/') - .filter(|s| !s.is_empty()) - .map(|s| s.to_string()) - .collect() -} - -pub fn item_parent_path(input: &str) -> Option { - let segs = item_path_segments(input); - if segs.len() <= 1 { - return None; - } - Some(segs[..segs.len() - 1].join("/")) -} +pub use slug_types::paths::{canonicalize_item, canonicalize_tag, item_parent_path, item_path_segments}; diff --git a/server/src/html/forum.rs b/server/src/html/forum.rs index ab7970feca47a4431ddc19dab2c8186180b78413..f789e347dc136f8ffd356f4492f0bd76cea04bf0 100644 --- a/server/src/html/forum.rs +++ b/server/src/html/forum.rs @@ -697,7 +697,7 @@ async fn thread_view_inner( let offset = q.offset.unwrap_or(0); let page_ids: Vec = all_ids.into_iter().skip(offset).take(PAGE_SIZE).collect(); - let (display_ingests, subtitle) = { + let (display_ingests, _subtitle) = { let reduced = state.reduced.read().await; let ingests = page_ids .iter() diff --git a/server/src/path_types.rs b/server/src/path_types.rs index cad58cac7da948f92d6ea2a5587d917ab21d57ea..4c8075bbd488f1b9a5eda9e03ced83f6238c6d5e 100644 --- a/server/src/path_types.rs +++ b/server/src/path_types.rs @@ -1,260 +1,3 @@ -//! Path representation types. -//! -//! The codebase currently treats item identifiers as strings in a few different -//! encodings: -//! - user/DSL input like `~/a/b` -//! - canonical item URLs like `https://slug.social/~/a/b` -//! - relative paths within a rooted tree view (e.g. `llms/openai` under a root) -//! -//! This module adds lightweight newtypes so code can be explicit about what it -//! expects without changing core storage formats. -//! -//! **Storage vs wire:** [`CanonicalItemUrl`] values are shared across scopes -//! (`https://slug.social/~/…`); which [`crate::reducer::ContentState`] they live in -//! is determined by scope, not by embedding the room id in the string. For JSON/RPC -//! and browser links in a private room, use [`crate::api::helpers::item_path_for_api_in_room`] -//! so ontology items become `https://slug.social/r/{short}/{slug}/~/…`. +//! Re-exports — implementations live in `slug-types` (`paths` module). -use std::borrow::Borrow; -use std::fmt; - -use serde::{Deserialize, Serialize}; - -use crate::canonical_path::canonicalize_item; - -/// Canonical item identifier as produced by `canonical_path::canonicalize_item`. -/// -/// In practice this is usually: -/// - `https://slug.social/~/...` for ontology items, or -/// - `https://...` / `http://...` for URL items. -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -pub struct CanonicalItemUrl(pub String); - -impl CanonicalItemUrl { - pub fn parse(input: &str) -> Option { - let c = canonicalize_item(input); - if c.is_empty() { - None - } else { - Some(Self(c)) - } - } - - pub fn as_str(&self) -> &str { - &self.0 - } - - /// Returns the `~/...` tail for ontology items (`https://slug.social/~/...`). - pub fn tilde_tail(&self) -> Option<&str> { - self.0.strip_prefix("https://slug.social/~/") - } - - /// Returns the final non-empty `/`-separated segment of the path. - /// - /// `https://slug.social/~/a/b/c` → `"c"` - /// `https://slug.social/~/a` → `"a"` - pub fn last_segment(&self) -> &str { - self.0 - .rsplit('/') - .find(|s| !s.is_empty()) - .unwrap_or(self.0.as_str()) - } - - /// The ontology root key as stored in `item_children`: `"https://slug.social/~"`. - /// Use this (not `parse("~/")`) when looking up top-level children. - pub fn ontology_root() -> Self { - Self("https://slug.social/~".to_string()) - } - - /// Returns the parent of this canonical item URL by stripping the last - /// path segment, or `None` if there is no parent (already at root). - /// - /// `https://slug.social/~/a/b/c` → `Some("https://slug.social/~/a/b")` - /// `https://slug.social/~/a` → `Some("https://slug.social/~")` - /// `https://slug.social/~/` → `None` (tilde root) - pub fn parent(&self) -> Option { - // tilde_tail() is None for non-ontology URLs and "" for the root ~/ - if self.tilde_tail().map(|t| t.is_empty()).unwrap_or(true) { - return None; - } - // Strip everything from the last '/' onwards. - let last_slash = self.0.rfind('/')?; - let parent_str = &self.0[..last_slash]; - if parent_str.is_empty() { - None - } else { - Some(Self(parent_str.to_string())) - } - } - - /// Segments of an ontology path suitable for breadcrumb rendering. - /// Strips the `https://slug.social` prefix and returns the `~/…` parts. - /// - /// `https://slug.social/~/a/b` → `["~", "a", "b"]` - /// `https://slug.social/~/` → `["~"]` - pub fn tilde_segments(&self) -> Vec<&str> { - match self.tilde_tail() { - Some(tail) if !tail.is_empty() => { - std::iter::once("~") - .chain(tail.split('/').filter(|s| !s.is_empty())) - .collect() - } - Some(_) => vec!["~"], - None => vec![], - } - } -} - -impl fmt::Display for CanonicalItemUrl { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - self.0.fmt(f) - } -} - -/// Allow `HashMap` to be searched by `&str`. -impl Borrow for CanonicalItemUrl { - fn borrow(&self) -> &str { - &self.0 - } -} - -impl PartialEq for CanonicalItemUrl { - fn eq(&self, other: &str) -> bool { - self.0 == other - } -} - -impl PartialEq<&str> for CanonicalItemUrl { - fn eq(&self, other: &&str) -> bool { - self.0 == *other - } -} - -impl PartialEq for CanonicalItemUrl { - fn eq(&self, other: &String) -> bool { - &self.0 == other - } -} - -/// A `~/...` input path (as used in the DSL and UX). -/// -/// This is not canonicalized; it is a presentation/input form. -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -pub struct TildePath(pub String); - -impl TildePath { - pub fn new(input: &str) -> Option { - let s = input.trim(); - if s.starts_with("~/") && s.len() > 2 { - Some(Self(s.to_string())) - } else if s == "~/" { - Some(Self("~/".to_string())) - } else { - None - } - } - - pub fn as_str(&self) -> &str { - &self.0 - } - - pub fn canonicalize(&self) -> Option { - CanonicalItemUrl::parse(&self.0) - } -} - -impl fmt::Display for TildePath { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - self.0.fmt(f) - } -} - -/// A path relative to a chosen root in a tree UI. -/// -/// This is intended for compact state encodings (blobs). It must be joined to a -/// root `CanonicalItemUrl` (typically an ontology root) to become a full item. -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -pub struct RelativePath(pub String); - -impl RelativePath { - pub fn new(input: &str) -> Option { - let s = input.trim().trim_matches('/'); - if s.is_empty() { - Some(Self(String::new())) - } else { - // Keep this permissive: the DSL parser is the main gatekeeper. - Some(Self(s.to_string())) - } - } - - pub fn as_str(&self) -> &str { - &self.0 - } - - /// Join this relative path under a canonical ontology root - /// (`https://slug.social/~/...`) to form a canonical item URL. - pub fn join_under_ontology_root(&self, root: &CanonicalItemUrl) -> Option { - let base = root.tilde_tail()?; - // base is the tail after https://slug.social/~/, e.g. "models" or "models/llms" - let joined = if base.is_empty() { - if self.0.is_empty() { - "~/".to_string() - } else { - format!("~/{}", self.0) - } - } else if self.0.is_empty() { - format!("~/{}", base) - } else { - format!("~/{}/{}", base.trim_end_matches('/'), self.0) - }; - CanonicalItemUrl::parse(&joined) - } -} - -impl fmt::Display for RelativePath { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - self.0.fmt(f) - } -} - - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn canonical_parent_deep() { - let c = CanonicalItemUrl::parse("~/a/b/c").unwrap(); - assert_eq!(c.parent().unwrap().as_str(), "https://slug.social/~/a/b"); - } - - #[test] - fn canonical_parent_one_level() { - let c = CanonicalItemUrl::parse("~/a").unwrap(); - assert_eq!(c.parent().unwrap().as_str(), "https://slug.social/~"); - } - - #[test] - fn canonical_parent_root_is_none() { - let root = CanonicalItemUrl::parse("~/").unwrap(); - assert!(root.parent().is_none()); - } - - #[test] - fn tilde_segments_deep() { - let c = CanonicalItemUrl::parse("~/a/b").unwrap(); - assert_eq!(c.tilde_segments(), vec!["~", "a", "b"]); - } - - #[test] - fn tilde_segments_root() { - let c = CanonicalItemUrl::parse("~/").unwrap(); - assert_eq!(c.tilde_segments(), vec!["~"]); - } - - #[test] - fn tilde_segments_non_ontology_is_empty() { - let c = CanonicalItemUrl::parse("https://example.com/foo").unwrap(); - assert_eq!(c.tilde_segments(), Vec::<&str>::new()); - } -} +pub use slug_types::paths::{CanonicalItemUrl, RelativePath, TildePath}; diff --git a/types/src/lib.rs b/types/src/lib.rs index c1cde3b783d03b02783385b5f07659fb112aae3e..5fc867bf2af84c2ca63fa1cdd03413110ad784a3 100644 --- a/types/src/lib.rs +++ b/types/src/lib.rs @@ -1,7 +1,13 @@ use serde::{Deserialize, Serialize}; +pub mod paths; pub mod timeago; +pub use paths::{ + canonicalize_item, canonicalize_tag, item_parent_path, item_path_segments, CanonicalItemUrl, + ForumThreadUrl, GardenItemUrl, RelativePath, TildeOntologyPath, TildePath, +}; + #[derive(Debug, Serialize, Deserialize)] pub struct ApiError { pub ok: bool, @@ -12,7 +18,7 @@ pub struct ApiError { #[derive(Debug, Clone, Serialize, Deserialize)] pub struct RankRow { - pub item: String, + pub item: GardenItemUrl, pub score: f64, /// Normalized score as a percentage of the top item (0–100). Present when ?percent=true. #[serde(skip_serializing_if = "Option::is_none")] @@ -43,7 +49,7 @@ pub struct RankComponent { #[derive(Debug, Serialize, Deserialize)] pub struct RankResponse { pub components: Vec, - pub unranked_items: Vec, + pub unranked_items: Vec, } /// Graph connectivity stats for a scope, returned with pair suggestions. @@ -63,8 +69,8 @@ pub struct ConnectivityStats { #[derive(Debug, Serialize, Deserialize)] pub struct PairResponse { - pub left: String, - pub right: String, + pub left: GardenItemUrl, + pub right: GardenItemUrl, pub left_body: Option, pub right_body: Option, /// Thread tags that discuss either item (connective tissue to forum). @@ -79,7 +85,7 @@ pub struct PairResponse { pub struct NextMoves { pub pair: String, pub rank: String, - pub web: String, + pub web: ForumThreadUrl, } #[derive(Debug, Serialize, Deserialize)] @@ -90,14 +96,14 @@ pub struct PathsResponse { /// Leaf items only (no children). For search / "full path list" — does not scale, works for now. #[derive(Debug, Serialize, Deserialize)] pub struct LeavesResponse { - pub paths: Vec, + pub paths: Vec, } #[derive(Debug, Serialize, Deserialize)] pub struct PathSummary { - pub path: String, + pub path: TildeOntologyPath, pub children: usize, - pub web: String, + pub web: GardenItemUrl, } #[derive(Debug, Clone, Serialize, Deserialize)] @@ -109,7 +115,7 @@ pub struct ThreadsResponse { pub struct ThreadSummary { pub thread: String, pub last_activity_ts: i64, - pub web: String, + pub web: ForumThreadUrl, } #[derive(Debug, Serialize, Deserialize)] @@ -178,7 +184,7 @@ pub struct IngestRow { #[derive(Debug, Serialize, Deserialize)] pub struct ItemResponse { - pub item: String, + pub item: GardenItemUrl, pub body: Option, /// True when the body was truncated due to size. Fetch with `?full=true` for the complete body. #[serde(default, skip_serializing_if = "std::ops::Not::not")] @@ -201,15 +207,15 @@ pub struct RecentVotesResponse { /// Vote history for one item (matchup: wins/losses + thread per vote). #[derive(Debug, Serialize, Deserialize)] pub struct MatchupResponse { - pub item: String, + pub item: GardenItemUrl, pub votes: Vec, } #[derive(Debug, Clone, Serialize, Deserialize)] pub struct VoteRow { pub ts: i64, - pub a: String, - pub b: String, + pub a: GardenItemUrl, + pub b: GardenItemUrl, pub ratio: String, /// Principal username when present (stored form, no `@`). pub actor: Option, @@ -510,7 +516,7 @@ pub struct RankPosition { /// How one item's rank changed after a vote. #[derive(Debug, Clone, Serialize, Deserialize)] pub struct RankChange { - pub item: String, + pub item: GardenItemUrl, /// Position before the vote. None = was unranked (no voted connections in this scope). pub before: Option, /// Position after the vote. None = became unranked (e.g. component split, unlikely). @@ -544,7 +550,7 @@ pub struct CheckScopeRanking { /// Parent scope path (e.g. "/models" or "/" for root). pub parent: String, pub components: Vec, - pub unranked_items: Vec, + pub unranked_items: Vec, } #[derive(Debug, Serialize, Deserialize)] @@ -567,7 +573,7 @@ pub struct SearchResponse { #[derive(Debug, Serialize, Deserialize)] pub struct SearchItemHit { - pub path: String, + pub path: GardenItemUrl, #[serde(skip_serializing_if = "Option::is_none")] pub body: Option, } @@ -614,7 +620,7 @@ pub struct RankHistoryRow { #[derive(Debug, Serialize, Deserialize)] pub struct RankHistoryResponse { - pub item: String, + pub item: GardenItemUrl, pub history: Vec, } diff --git a/types/src/paths.rs b/types/src/paths.rs new file mode 100644 index 0000000000000000000000000000000000000000..2950a7502255927583fbacdfd2adb700f0b0c221 --- /dev/null +++ b/types/src/paths.rs @@ -0,0 +1,498 @@ +//! Canonical paths, storage ids, and JSON href newtypes. All normalization and +//! room-aware URL rules for items live here. + +use std::borrow::Borrow; +use std::fmt; + +use serde::{Deserialize, Serialize}; + +// --------------------------------------------------------------------------- +// Normalization (moved from server `canonical_path`) +// --------------------------------------------------------------------------- + +/// Thread / public tag: stored without leading `#`, lowercase. +pub fn canonicalize_tag(input: &str) -> String { + input.trim().trim_start_matches('#').to_lowercase() +} + +/// Ontology item reference → canonical absolute URL on the slug host. +pub fn canonicalize_item(input: &str) -> String { + let s = input.trim(); + if s.is_empty() { + return String::new(); + } + + if let Some(rest) = s.strip_prefix("https://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let host = host.trim().to_lowercase(); + if tail.is_empty() { + return format!("https://{}", host); + } else { + return format!("https://{}/{}", host, tail); + } + } + if let Some(rest) = s.strip_prefix("http://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let host = host.trim().to_lowercase(); + if tail.is_empty() { + return format!("http://{}", host); + } else { + return format!("http://{}/{}", host, tail); + } + } + + let is_tilde = s.starts_with("~/"); + let rest = s.strip_prefix("~/").or_else(|| s.strip_prefix("/")).unwrap_or(s); + + let tail = rest + .split('/') + .filter_map(|seg| { + let t = seg.trim(); + if t.is_empty() { + None + } else { + Some(t.to_lowercase()) + } + }) + .collect::>() + .join("/"); + + if is_tilde { + format!("https://slug.social/~/{}", tail) + } else if tail.is_empty() { + "https://slug.social".to_string() + } else { + format!("https://slug.social/{}", tail) + } +} + +pub fn item_path_segments(input: &str) -> Vec { + let canonical = canonicalize_item(input); + if canonical.is_empty() { + return vec![]; + } + + if let Some(rest) = canonical.strip_prefix("https://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let mut out = vec![format!("https://{}", host)]; + out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); + return out; + } + if let Some(rest) = canonical.strip_prefix("http://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let mut out = vec![format!("http://{}", host)]; + out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); + return out; + } + + canonical + .split('/') + .filter(|s| !s.is_empty()) + .map(|s| s.to_string()) + .collect() +} + +pub fn item_parent_path(input: &str) -> Option { + let segs = item_path_segments(input); + if segs.len() <= 1 { + return None; + } + Some(segs[..segs.len() - 1].join("/")) +} + +// --------------------------------------------------------------------------- +// Storage + input path newtypes +// --------------------------------------------------------------------------- + +/// Canonical item identifier as produced by [`canonicalize_item`]. +/// +/// Shared across all scopes; room is not embedded. Usually +/// `https://slug.social/~/…` or an external `http(s)://…` URL item. +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +pub struct CanonicalItemUrl(pub String); + +impl CanonicalItemUrl { + pub fn parse(input: &str) -> Option { + let c = canonicalize_item(input); + if c.is_empty() { + None + } else { + Some(Self(c)) + } + } + + pub fn as_str(&self) -> &str { + &self.0 + } + + pub fn tilde_tail(&self) -> Option<&str> { + self.0.strip_prefix("https://slug.social/~/") + } + + pub fn last_segment(&self) -> &str { + self.0 + .rsplit('/') + .find(|s| !s.is_empty()) + .unwrap_or(self.0.as_str()) + } + + pub fn ontology_root() -> Self { + Self("https://slug.social/~".to_string()) + } + + pub fn parent(&self) -> Option { + if self.tilde_tail().map(|t| t.is_empty()).unwrap_or(true) { + return None; + } + let last_slash = self.0.rfind('/')?; + let parent_str = &self.0[..last_slash]; + if parent_str.is_empty() { + None + } else { + Some(Self(parent_str.to_string())) + } + } + + pub fn tilde_segments(&self) -> Vec<&str> { + match self.tilde_tail() { + Some(tail) if !tail.is_empty() => { + std::iter::once("~") + .chain(tail.split('/').filter(|s| !s.is_empty())) + .collect() + } + Some(_) => vec!["~"], + None => vec![], + } + } + + /// `~/…` list label for ontology items (paths index, CLI). + pub fn tilde_list_label(&self) -> TildeOntologyPath { + TildeOntologyPath::from_stored(self) + } + + /// Absolute href for JSON/RPC and browsers for this stored id in `room`. + pub fn json_href(&self, room_wire: &str) -> GardenItemUrl { + GardenItemUrl::from_stored(self, room_wire) + } +} + +impl fmt::Display for CanonicalItemUrl { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +impl Borrow for CanonicalItemUrl { + fn borrow(&self) -> &str { + &self.0 + } +} + +impl PartialEq for CanonicalItemUrl { + fn eq(&self, other: &str) -> bool { + self.0 == other + } +} + +impl PartialEq<&str> for CanonicalItemUrl { + fn eq(&self, other: &&str) -> bool { + self.0 == *other + } +} + +impl PartialEq for CanonicalItemUrl { + fn eq(&self, other: &String) -> bool { + &self.0 == other + } +} + +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +pub struct TildePath(pub String); + +impl TildePath { + pub fn new(input: &str) -> Option { + let s = input.trim(); + if s.starts_with("~/") && s.len() > 2 { + Some(Self(s.to_string())) + } else if s == "~/" { + Some(Self("~/".to_string())) + } else { + None + } + } + + pub fn as_str(&self) -> &str { + &self.0 + } + + pub fn canonicalize(&self) -> Option { + CanonicalItemUrl::parse(&self.0) + } +} + +impl fmt::Display for TildePath { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +pub struct RelativePath(pub String); + +impl RelativePath { + pub fn new(input: &str) -> Option { + let s = input.trim().trim_matches('/'); + if s.is_empty() { + Some(Self(String::new())) + } else { + Some(Self(s.to_string())) + } + } + + pub fn as_str(&self) -> &str { + &self.0 + } + + pub fn join_under_ontology_root(&self, root: &CanonicalItemUrl) -> Option { + let base = root.tilde_tail()?; + let joined = if base.is_empty() { + if self.0.is_empty() { + "~/".to_string() + } else { + format!("~/{}", self.0) + } + } else if self.0.is_empty() { + format!("~/{}", base) + } else { + format!("~/{}/{}", base.trim_end_matches('/'), self.0) + }; + CanonicalItemUrl::parse(&joined) + } +} + +impl fmt::Display for RelativePath { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +// --------------------------------------------------------------------------- +// Wire / JSON: correct-by-construction hrefs +// --------------------------------------------------------------------------- + +fn api_path_or_url(item: &str) -> String { + if item.starts_with("http://") || item.starts_with("https://") { + item.to_string() + } else { + format!("/{}", item) + } +} + +/// Ontology item as serialized in JSON (absolute URL or `/`-prefixed path). +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(transparent)] +pub struct GardenItemUrl(pub String); + +impl GardenItemUrl { + pub fn as_str(&self) -> &str { + &self.0 + } + + pub fn into_inner(self) -> String { + self.0 + } + + /// Stored canonical id + RPC `room` field (`"public"` or `"short/slug"`). + pub fn from_stored(stored: &CanonicalItemUrl, room_wire: &str) -> Self { + Self(garden_href_string(stored.as_str(), room_wire)) + } + + /// Like [`Self::from_stored`] but accepts a string that may already be canonical. + pub fn from_storage_str(stored: &str, room_wire: &str) -> Self { + Self(garden_href_string(stored, room_wire)) + } +} + +impl fmt::Display for GardenItemUrl { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +fn garden_href_string(item: &str, room_wire: &str) -> String { + let room = room_wire.trim(); + if room.is_empty() || room == "public" { + return api_path_or_url(item); + } + let Some((short, slug)) = room.split_once('/') else { + return api_path_or_url(item); + }; + if short.is_empty() || slug.is_empty() { + return api_path_or_url(item); + } + let Some(c) = CanonicalItemUrl::parse(item) else { + return api_path_or_url(item); + }; + let root = CanonicalItemUrl::ontology_root(); + let item_norm = c.as_str().trim_end_matches('/'); + let root_norm = root.as_str().trim_end_matches('/'); + if let Some(tail) = c.tilde_tail() { + return if tail.is_empty() { + format!("https://slug.social/r/{short}/{slug}/~") + } else { + format!("https://slug.social/r/{short}/{slug}/~/{}", tail) + }; + } + if item_norm == root_norm { + return format!("https://slug.social/r/{short}/{slug}/~"); + } + api_path_or_url(item) +} + +/// Forum thread URL for JSON (`/t/…` or `/r/…/t/…` on slug.social). +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(transparent)] +pub struct ForumThreadUrl(pub String); + +impl ForumThreadUrl { + pub fn as_str(&self) -> &str { + &self.0 + } + + pub fn into_inner(self) -> String { + self.0 + } + + pub fn from_room_tag(room_wire: &str, thread_tag: &str) -> Self { + let room = room_wire.trim(); + let tag = thread_tag.trim().trim_start_matches('#'); + Self(if room.is_empty() || room == "public" { + format!("https://slug.social/t/{tag}") + } else if let Some((short, slug)) = room.split_once('/') { + if short.is_empty() || slug.is_empty() { + format!("https://slug.social/t/{tag}") + } else { + format!("https://slug.social/r/{short}/{slug}/t/{tag}") + } + } else { + format!("https://slug.social/t/{tag}") + }) + } +} + +impl fmt::Display for ForumThreadUrl { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +/// `~/a/b` style path for list UIs (paths index `path` field). +#[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(transparent)] +pub struct TildeOntologyPath(pub String); + +impl TildeOntologyPath { + pub fn from_stored(c: &CanonicalItemUrl) -> Self { + let s = match c.tilde_tail() { + Some(tail) if !tail.is_empty() => format!("~/{}", tail), + Some(_) => "~/".to_string(), + None => c.to_string(), + }; + Self(s) + } + + pub fn as_str(&self) -> &str { + &self.0 + } +} + +impl fmt::Display for TildeOntologyPath { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn canonical_parent_deep() { + let c = CanonicalItemUrl::parse("~/a/b/c").unwrap(); + assert_eq!(c.parent().unwrap().as_str(), "https://slug.social/~/a/b"); + } + + #[test] + fn canonical_parent_one_level() { + let c = CanonicalItemUrl::parse("~/a").unwrap(); + assert_eq!(c.parent().unwrap().as_str(), "https://slug.social/~"); + } + + #[test] + fn canonical_parent_root_is_none() { + let root = CanonicalItemUrl::parse("~/").unwrap(); + assert!(root.parent().is_none()); + } + + #[test] + fn tilde_segments_deep() { + let c = CanonicalItemUrl::parse("~/a/b").unwrap(); + assert_eq!(c.tilde_segments(), vec!["~", "a", "b"]); + } + + #[test] + fn tilde_segments_root() { + let c = CanonicalItemUrl::parse("~/").unwrap(); + assert_eq!(c.tilde_segments(), vec!["~"]); + } + + #[test] + fn tilde_segments_non_ontology_is_empty() { + let c = CanonicalItemUrl::parse("https://example.com/foo").unwrap(); + assert_eq!(c.tilde_segments(), Vec::<&str>::new()); + } + + #[test] + fn garden_public_passthrough_https() { + let u = "https://slug.social/~/a/b"; + assert_eq!(GardenItemUrl::from_storage_str(u, "public").as_str(), u); + } + + #[test] + fn garden_private_room_prefixes_ontology() { + assert_eq!( + GardenItemUrl::from_storage_str("https://slug.social/~/topic/x", "9ab12cd/my-room").as_str(), + "https://slug.social/r/9ab12cd/my-room/~/topic/x" + ); + } + + #[test] + fn garden_private_room_ontology_root() { + assert_eq!( + GardenItemUrl::from_storage_str("https://slug.social/~", "9ab12cd/my-room").as_str(), + "https://slug.social/r/9ab12cd/my-room/~" + ); + assert_eq!( + GardenItemUrl::from_storage_str("https://slug.social/~/", "9ab12cd/my-room").as_str(), + "https://slug.social/r/9ab12cd/my-room/~" + ); + } + + #[test] + fn garden_external_url_untouched_in_private_room() { + let u = "https://example.com/z"; + assert_eq!(GardenItemUrl::from_storage_str(u, "9ab12cd/my-room").as_str(), u); + } + + #[test] + fn forum_web_public_vs_room() { + assert_eq!( + ForumThreadUrl::from_room_tag("public", "debate").as_str(), + "https://slug.social/t/debate" + ); + assert_eq!( + ForumThreadUrl::from_room_tag("9ab12cd/my-room", "#debate").as_str(), + "https://slug.social/r/9ab12cd/my-room/t/debate" + ); + } +} Side B — contributor: tommy-mor Side B — commit message: [6e344666] Improve sorterc scan speed, errors, and ingest compile. Make scan a fast DSL parse pass with human-readable output, surface full parse_error details, and add compile --ingest for single-event replay from a log. Co-authored-by: Cursor Side B — unified diff (full patch): diff --git a/server/src/offline.rs b/server/src/offline.rs index 54ad0ded096a305ef8454ab2cdd1c3af71b14f5d..2db6f67e22e00a24ce673d13095644dd9fa9342d 100644 --- a/server/src/offline.rs +++ b/server/src/offline.rs @@ -28,6 +28,10 @@ pub struct CompileResult { pub threads: Vec, pub rankings: Vec, pub stats: CompileStats, + #[serde(skip_serializing_if = "Option::is_none")] + pub ingest_id: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub ingest_line: Option, } #[derive(Debug, Clone, Serialize)] @@ -36,6 +40,8 @@ pub struct CompileError { pub error: String, #[serde(skip_serializing_if = "Option::is_none")] pub hint: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub parse_error: Option, } #[derive(Debug, Clone, Serialize)] @@ -50,7 +56,7 @@ pub struct MalformedIngest { pub id: String, pub room_id: String, pub thread_tag: String, - pub reason: String, + pub parse_error: String, } #[derive(Debug, Clone, Serialize)] @@ -59,9 +65,36 @@ pub struct ScanResult { pub path: String, pub total_lines: usize, pub parsed_events: usize, + pub ingest_events: usize, pub bad_json_lines: Vec, pub malformed_ingests: Vec, - pub skipped_ingests: usize, +} + +#[derive(Debug)] +pub enum CompileIngestError { + NotFound(String), + Io(std::io::Error), + Compile(CompileError), +} + +impl CompileIngestError { + pub fn into_compile_error(self) -> CompileError { + match self { + Self::NotFound(id) => CompileError { + ok: false, + error: format!("ingest not found: {id}"), + hint: Some("pass the ingest event id from events.jsonl".into()), + parse_error: None, + }, + Self::Io(e) => CompileError { + ok: false, + error: format!("io error: {e}"), + hint: None, + parse_error: None, + }, + Self::Compile(e) => e, + } + } } fn document_stats(doc: &dsl::Document) -> CompileStats { @@ -166,19 +199,22 @@ fn rankings_for_simulated( .collect() } -/// Validate and simulate one `.sorter` document against optional base reducer state. -pub fn compile_document( +fn compile_document_inner( base: &ReducerState, room: &str, text: &str, + ingest_id: Option, + ingest_line: Option, ) -> Result { let room_key = room.trim(); let scope = scope_from_room_wire(room_key); let validated = validate_ingest_document(base, text, &scope).map_err(|(_, message, hint)| { + let parse_error = dsl::parse_full(text).err().map(|e| e.to_string()); CompileError { ok: false, error: message, hint, + parse_error, } })?; @@ -200,15 +236,27 @@ pub fn compile_document( threads: threads_in_document(text), rankings: rankings_for_simulated(&simulated, &scope, room_key, &validated.doc), stats: document_stats(&validated.doc), + ingest_id, + ingest_line, }) } +/// Validate and simulate one `.sorter` document against optional base reducer state. +pub fn compile_document( + base: &ReducerState, + room: &str, + text: &str, +) -> Result { + compile_document_inner(base, room, text, None, None) +} + fn ingest_parse_error(raw: &str) -> Option { dsl::parse_full(raw).err().map(|e| e.to_string()) } -fn load_events_from_jsonl(path: &Path) -> Result<(Vec<(usize, Event)>, Vec), std::io::Error> { +fn load_events_from_jsonl(path: &Path) -> Result<(usize, Vec<(usize, Event)>, Vec), std::io::Error> { let text = std::fs::read_to_string(path)?; + let total_lines = text.lines().count(); let mut events = Vec::new(); let mut bad_json_lines = Vec::new(); for (idx, line) in text.lines().enumerate() { @@ -225,61 +273,90 @@ fn load_events_from_jsonl(path: &Path) -> Result<(Vec<(usize, Event)>, Vec Result<(ReducerState, Vec), std::io::Error> { - let (events, bad_json_lines) = load_events_from_jsonl(path)?; +fn replay_events(events: &[(usize, Event)]) -> ReducerState { let mut state = ReducerState::default(); for (_line_no, ev) in events { - state.apply_event(ev); + state.apply_event(ev.clone()); } - Ok((state, bad_json_lines)) + state } -/// Scan an events.jsonl for corrupt JSON lines and ingests that fail DSL replay. -pub fn scan_jsonl(path: &Path) -> Result { - let text = std::fs::read_to_string(path)?; - let total_lines = text.lines().count(); - let (events, bad_json_lines) = load_events_from_jsonl(path)?; +/// Replay a JSONL event log into reducer state (same rules as server boot). +pub fn load_reducer_from_jsonl(path: &Path) -> Result<(ReducerState, Vec), std::io::Error> { + let (_total_lines, events, bad_json_lines) = load_events_from_jsonl(path)?; + Ok((replay_events(&events), bad_json_lines)) +} - let mut malformed_ingests = Vec::new(); - let mut skipped_ingests = 0usize; - let mut state = ReducerState::default(); - let parsed_events = events.len(); +/// Find one ingest in a log and compile it against all prior events as base state. +pub fn compile_ingest_from_log(path: &Path, ingest_id: &str) -> Result { + let (_total_lines, events, bad_json_lines) = load_events_from_jsonl(path).map_err(CompileIngestError::Io)?; + if !bad_json_lines.is_empty() { + return Err(CompileIngestError::Compile(CompileError { + ok: false, + error: format!("jsonl has {} corrupt line(s)", bad_json_lines.len()), + hint: Some("fix the log or use `sorterc scan`".into()), + parse_error: None, + })); + } + + let needle = ingest_id.trim(); + let mut found: Option<(usize, Ingest)> = None; + let mut prior: Vec<(usize, Event)> = Vec::new(); for (line_no, ev) in events { if let Event::Ingest(ref ing) = ev { - if let Some(reason) = ingest_parse_error(&ing.raw) { + if ing.id == needle { + found = Some((line_no, ing.clone())); + break; + } + } + prior.push((line_no, ev)); + } + + let (line_no, ing) = found.ok_or_else(|| CompileIngestError::NotFound(needle.to_string()))?; + let base = replay_events(&prior); + compile_document_inner(&base, &ing.room_id, &ing.raw, Some(ing.id.clone()), Some(line_no)) + .map_err(CompileIngestError::Compile) +} + +/// Scan an events.jsonl for corrupt JSON lines and ingests whose DSL fails to parse. +/// +/// This does not replay the log (which would run rank centrality on every ingest and +/// can take minutes on real logs). It matches what the server skips on boot: parse failure. +pub fn scan_jsonl(path: &Path) -> Result { + let (total_lines, events, bad_json_lines) = load_events_from_jsonl(path)?; + + let mut malformed_ingests = Vec::new(); + let mut ingest_events = 0usize; + + for (line_no, ev) in &events { + if let Event::Ingest(ing) = ev { + ingest_events += 1; + if let Some(parse_error) = ingest_parse_error(&ing.raw) { malformed_ingests.push(MalformedIngest { - line: line_no, + line: *line_no, id: ing.id.clone(), room_id: ing.room_id.clone(), thread_tag: ing.thread_tag.clone(), - reason, + parse_error, }); } - let before = state.ingests_by_id.len(); - state.apply_event(ev); - if state.ingests_by_id.len() == before { - skipped_ingests += 1; - } - } else { - state.apply_event(ev); } } - let ok = bad_json_lines.is_empty() && malformed_ingests.is_empty() && skipped_ingests == 0; + let ok = bad_json_lines.is_empty() && malformed_ingests.is_empty(); Ok(ScanResult { ok, path: path.display().to_string(), total_lines, - parsed_events, + parsed_events: events.len(), + ingest_events, bad_json_lines, malformed_ingests, - skipped_ingests, }) } @@ -330,4 +407,25 @@ mod tests { assert!(!report.ok); assert_eq!(report.bad_json_lines.len(), 1); } + + #[test] + fn scan_reports_dsl_parse_error_detail() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("events.jsonl"); + let ingest = serde_json::json!({ + "type": "ingest", + "ts": 1, + "id": "bad-ingest-id", + "raw": "{ no closing brace\n~/a { body }\n~/a 1:0 ~/b", + "principal": "test", + "room_id": "public", + "thread_tag": "t", + }); + std::fs::write(&path, format!("{ingest}\n")).unwrap(); + let report = scan_jsonl(&path).unwrap(); + assert!(!report.ok); + assert_eq!(report.malformed_ingests.len(), 1); + assert_eq!(report.malformed_ingests[0].id, "bad-ingest-id"); + assert!(report.malformed_ingests[0].parse_error.contains("parse error")); + } } diff --git a/sorterc/readme.md b/sorterc/readme.md index 1ebcc3fc935541ea9e47e0458ec67a750fe19fff..e388615c34b40914ca51fc7a79cb738eebc2909b 100644 --- a/sorterc/readme.md +++ b/sorterc/readme.md @@ -60,21 +60,33 @@ Rankings use the same structure as the server's dry-run check: parent scope, con ### `scan` — lint an `events.jsonl` -Reads a JSONL event log and reports problems without starting a server. +Fast single-pass check. Does **not** replay the log (replay runs rank centrality on every ingest and gets slow fast). ```bash cargo run -p sorterc -- scan events.jsonl -cargo run -p sorterc -- scan events.jsonl --pretty +cargo run -p sorterc -- scan events.jsonl --json +cargo run -p sorterc -- scan events.jsonl --json --pretty ``` +Default output is human-readable with a blank line between each problem. Use `--json` for machine output. + Reports: - **bad JSON lines** — lines that are not valid JSON -- **malformed ingests** — ingest events whose `raw` DSL fails to parse -- **skipped ingests** — ingests dropped during replay (same behavior as server boot) +- **malformed ingests** — ingest events whose `raw` DSL fails to parse, with full `parse_error` text Exits 0 when clean, 1 when any issue is found. +### `compile --ingest` — compile one event from a log + +Replay all events **before** the target ingest as base state, then compile that ingest's DSL: + +```bash +cargo run -p sorterc -- compile --ingest cabd8adc-57ae-402d-a940-8e24339ac451 --from events.jsonl --pretty +``` + +Output includes `ingest_id`, `ingest_line`, and rankings for that post only. This may take a while for ingests late in a large log (full replay up to that point). + ## Typical uses - Iterate on `.sorter` files in an editor and pipe through `compile` to see rankings instantly diff --git a/sorterc/src/main.rs b/sorterc/src/main.rs index 71382c180085cb0ad71043c852f8db5d3a48a284..5687c850d16973d28749380b8dc89f3bcebbfc56 100644 --- a/sorterc/src/main.rs +++ b/sorterc/src/main.rs @@ -22,9 +22,15 @@ struct Cli { enum Command { /// Parse and simulate a .sorter document; emit ranking JSON to stdout. Compile { - /// `.sorter` file, or `-` for stdin. - file: PathBuf, - /// Room wire id (`public` or private room id). + /// `.sorter` file, or `-` for stdin. Omit when using --ingest. + file: Option, + /// Compile one ingest event from a log (by event id / uuid). + #[arg(long)] + ingest: Option, + /// events.jsonl containing the ingest (required with --ingest). + #[arg(long)] + from: Option, + /// Room wire id (`public` or private room id). Ignored with --ingest. #[arg(long, default_value = "public")] room: String, /// Optional events.jsonl to replay before compiling (seed garden state). @@ -34,9 +40,13 @@ enum Command { #[arg(long)] pretty: bool, }, - /// Scan an events.jsonl for corrupt JSON lines and malformed ingests. + /// Scan an events.jsonl for corrupt JSON lines and DSL parse failures. Scan { file: PathBuf, + /// Emit JSON instead of human-readable output. + #[arg(long)] + json: bool, + /// Pretty-print JSON (requires --json). #[arg(long)] pretty: bool, }, @@ -58,6 +68,7 @@ fn load_base_state(base: Option<&Path>) -> Result { let Some(path) = base else { return Ok(ReducerState::default()); }; + eprintln!("sorterc: replaying {} for base state (may take a while on large logs)…", path.display()); let (state, bad_lines) = offline::load_reducer_from_jsonl(path) .with_context(|| format!("load base jsonl {}", path.display()))?; if !bad_lines.is_empty() { @@ -78,25 +89,101 @@ fn print_json(value: &T, pretty: bool) -> Result<()> { Ok(()) } -fn run_compile(file: PathBuf, room: String, base: Option, pretty: bool) -> Result<()> { - let text = read_input(&file)?; - let base_state = load_base_state(base.as_deref())?; - match offline::compile_document(&base_state, &room, &text) { - Ok(result) => { - print_json::(&result, pretty)?; - Ok(()) +fn run_compile( + file: Option, + ingest: Option, + from: Option, + room: String, + base: Option, + pretty: bool, +) -> Result<()> { + if ingest.is_some() ^ from.is_some() { + bail!("--ingest and --from must be used together"); + } + if ingest.is_some() && (file.is_some() || base.is_some()) { + bail!("with --ingest/--from, omit file and --base"); + } + if file.is_none() && ingest.is_none() { + bail!("pass a .sorter file or --ingest --from events.jsonl"); + } + + if let (Some(ingest_id), Some(log_path)) = (ingest, from) { + eprintln!( + "sorterc: compiling ingest {ingest_id} from {}…", + log_path.display() + ); + match offline::compile_ingest_from_log(&log_path, &ingest_id) { + Ok(result) => { + print_json::(&result, pretty)?; + Ok(()) + } + Err(err) => { + print_json::(&err.into_compile_error(), pretty)?; + std::process::exit(1); + } + } + } else { + let file = file.expect("checked above"); + let text = read_input(&file)?; + let base_state = load_base_state(base.as_deref())?; + match offline::compile_document(&base_state, &room, &text) { + Ok(result) => { + print_json::(&result, pretty)?; + Ok(()) + } + Err(err) => { + print_json::(&err, pretty)?; + std::process::exit(1); + } + } + } +} + +fn print_scan_human(report: &ScanResult) { + if report.ok { + println!( + "ok: {} ({} lines, {} events, {} ingests)", + report.path, report.total_lines, report.parsed_events, report.ingest_events + ); + return; + } + + let problems = report.bad_json_lines.len() + report.malformed_ingests.len(); + println!( + "{}: {} lines, {} events, {} ingests, {problems} problem(s)", + report.path, report.total_lines, report.parsed_events, report.ingest_events + ); + + let mut first = true; + for bad in &report.bad_json_lines { + if !first { + println!(); } - Err(err) => { - print_json::(&err, pretty)?; - std::process::exit(1); + first = false; + println!("line {}: invalid JSON", bad.line); + println!("{}", bad.message); + } + + for bad in &report.malformed_ingests { + if !first { + println!(); } + first = false; + println!( + "line {}: ingest {} (#{} in {})", + bad.line, bad.id, bad.thread_tag, bad.room_id + ); + println!("{}", bad.parse_error); } } -fn run_scan(file: PathBuf, pretty: bool) -> Result<()> { - let report = offline::scan_jsonl(&file) - .with_context(|| format!("scan {}", file.display()))?; - print_json::(&report, pretty)?; +fn run_scan(file: PathBuf, json: bool, pretty: bool) -> Result<()> { + let report = offline::scan_jsonl(&file).with_context(|| format!("scan {}", file.display()))?; + if json { + print_json::(&report, pretty)?; + } else { + print_scan_human(&report); + } if !report.ok { std::process::exit(1); } @@ -108,10 +195,12 @@ fn main() -> Result<()> { match cli.cmd { Command::Compile { file, + ingest, + from, room, base, pretty, - } => run_compile(file, room, base, pretty), - Command::Scan { file, pretty } => run_scan(file, pretty), + } => run_compile(file, ingest, from, room, base, pretty), + Command::Scan { file, json, pretty } => run_scan(file, json, pretty), } }