You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [a888d56c] refactor: centralize path identity in slug-types Move canonicalization and CanonicalItemUrl into types::paths with GardenItemUrl, ForumThreadUrl, and TildeOntologyPath for JSON hrefs. Server canonical_path and path_types re-export slug-types; RPC and validation build hrefs via those types instead of string helpers. Made-with: Cursor Side A — unified diff (full patch): diff --git a/server/src/api/helpers.rs b/server/src/api/helpers.rs index 9b71491e9f9efc44a2a4beba09be8f64bd2ff2ee..03b3e77911ccd662bec8635345dafe2593cf242e 100644 --- a/server/src/api/helpers.rs +++ b/server/src/api/helpers.rs @@ -4,12 +4,12 @@ use axum::{ Json, }; use sha2::{Digest, Sha256}; +use slug_types::paths::{CanonicalItemUrl, GardenItemUrl}; use slug_types::*; use std::collections::HashMap; use crate::{ canonical_path::canonicalize_item, - path_types::CanonicalItemUrl, ranking::connected_components_from_voted_pairs, }; @@ -30,64 +30,6 @@ pub fn now_ms() -> i64 { t.as_millis() as i64 } -/// Serialize a canonical item for JSON: absolute URLs stay as-is; bare paths get a `/` prefix. -pub fn item_path_for_api(item: &str) -> String { - if item.starts_with("http://") || item.starts_with("https://") { - item.to_string() - } else { - format!("/{}", item) - } -} - -/// Same as [`item_path_for_api`], but for private rooms ontology items are prefixed with -/// `/r/{short}/{slug}` so the URL matches the web app (`/r/…/~/…` routes). -pub fn item_path_for_api_in_room(item: &str, room_wire: &str) -> String { - let room = room_wire.trim(); - if room.is_empty() || room == "public" { - return item_path_for_api(item); - } - let Some((short, slug)) = room.split_once('/') else { - return item_path_for_api(item); - }; - if short.is_empty() || slug.is_empty() { - return item_path_for_api(item); - } - let Some(c) = CanonicalItemUrl::parse(item) else { - return item_path_for_api(item); - }; - let root = CanonicalItemUrl::ontology_root(); - let item_norm = c.as_str().trim_end_matches('/'); - let root_norm = root.as_str().trim_end_matches('/'); - if let Some(tail) = c.tilde_tail() { - return if tail.is_empty() { - format!("https://slug.social/r/{short}/{slug}/~") - } else { - format!("https://slug.social/r/{short}/{slug}/~/{}", tail) - }; - } - if item_norm == root_norm { - return format!("https://slug.social/r/{short}/{slug}/~"); - } - item_path_for_api(item) -} - -/// Absolute thread URL for forum JSON (`/t/…` vs `/r/…/t/…`). -pub fn forum_thread_web_url(room_wire: &str, thread_tag: &str) -> String { - let room = room_wire.trim(); - let tag = thread_tag.trim().trim_start_matches('#'); - if room.is_empty() || room == "public" { - format!("https://slug.social/t/{tag}") - } else if let Some((short, slug)) = room.split_once('/') { - if short.is_empty() || slug.is_empty() { - format!("https://slug.social/t/{tag}") - } else { - format!("https://slug.social/r/{short}/{slug}/t/{tag}") - } - } else { - format!("https://slug.social/t/{tag}") - } -} - /// Resolve an item path as a first-class canonical path. pub fn resolve_item(item: &str) -> Result { let canonical = canonicalize_item(item); @@ -109,14 +51,12 @@ pub fn parse_parent_specs(parent: Option<&String>) -> Vec { } /// Apply offset+limit pagination to the flattened component rankings. -/// Items are flattened in component order (largest component first), then unranked last. -/// Returns (components, unranked_items) after the window. pub fn paginate_rankings( components: Vec, - unranked_items: Vec, + unranked_items: Vec, offset: usize, limit: Option, -) -> (Vec, Vec) { +) -> (Vec, Vec) { let mut remaining_skip = offset; let mut remaining_take = limit.unwrap_or(usize::MAX); let mut out_components: Vec = Vec::new(); @@ -141,7 +81,7 @@ pub fn paginate_rankings( }); } - let out_unranked: Vec = if remaining_take > 0 { + let out_unranked: Vec = if remaining_take > 0 { unranked_items .into_iter() .skip(remaining_skip) @@ -183,11 +123,9 @@ pub fn is_pair_voted(group: &crate::reducer::GroupState, a: &str, b: &str) -> bo group.voted_pairs.contains(&(i, j)) } -/// Compute graph connectivity stats for a set of items within the ranking group. pub fn compute_connectivity_stats(group: &crate::reducer::GroupState, pool: &[String]) -> ConnectivityStats { let n = pool.len(); - // Map pool items to global indices (items not yet in the group get no index) let global_idxs: Vec> = pool .iter() .map(|it| { @@ -197,7 +135,6 @@ pub fn compute_connectivity_stats(group: &crate::reducer::GroupState, pool: &[St .collect(); let present: Vec = global_idxs.iter().filter_map(|x| *x).collect(); - // Build local index mapping for items that exist in the ranking group let global_to_local: HashMap = present .iter() .enumerate() @@ -213,7 +150,6 @@ pub fn compute_connectivity_stats(group: &crate::reducer::GroupState, pool: &[St }), ); - // Items not in the ranking group at all are also isolates let items_not_in_group = global_idxs.iter().filter(|x| x.is_none()).count(); let num_components = comps.len() + isolates.len() + items_not_in_group; @@ -237,52 +173,3 @@ pub fn vote_touches_path(a: &str, b: &str, parent_canon: &str) -> bool { let under = |item: &str| item == parent_canon || item.starts_with(&format!("{}/", parent_canon)); under(a) || under(b) } - -#[cfg(test)] -mod wire_url_tests { - use super::{forum_thread_web_url, item_path_for_api_in_room}; - - #[test] - fn public_room_unchanged() { - let u = "https://slug.social/~/a/b"; - assert_eq!(item_path_for_api_in_room(u, "public"), u); - } - - #[test] - fn private_room_prefixes_ontology() { - assert_eq!( - item_path_for_api_in_room("https://slug.social/~/topic/x", "9ab12cd/my-room"), - "https://slug.social/r/9ab12cd/my-room/~/topic/x" - ); - } - - #[test] - fn private_room_ontology_root() { - assert_eq!( - item_path_for_api_in_room("https://slug.social/~", "9ab12cd/my-room"), - "https://slug.social/r/9ab12cd/my-room/~" - ); - assert_eq!( - item_path_for_api_in_room("https://slug.social/~/", "9ab12cd/my-room"), - "https://slug.social/r/9ab12cd/my-room/~" - ); - } - - #[test] - fn external_url_untouched_in_private_room() { - let u = "https://example.com/z"; - assert_eq!(item_path_for_api_in_room(u, "9ab12cd/my-room"), u); - } - - #[test] - fn forum_web_public_vs_room() { - assert_eq!( - forum_thread_web_url("public", "debate"), - "https://slug.social/t/debate" - ); - assert_eq!( - forum_thread_web_url("9ab12cd/my-room", "#debate"), - "https://slug.social/r/9ab12cd/my-room/t/debate" - ); - } -} diff --git a/server/src/api/mod.rs b/server/src/api/mod.rs index 042aa248305f9362a3be78f9eea2a5abf6ba707a..cf22cb0129366c3aed031bc86f3197a4321cb806 100644 --- a/server/src/api/mod.rs +++ b/server/src/api/mod.rs @@ -24,8 +24,7 @@ pub use auth::{ pub use helpers::{ api_error, compute_connectivity_stats, is_pair_voted, now_ms, paginate_rankings, - parse_parent_specs, pick_random_distinct, sha256_hex, resolve_item, vote_touches_path, - item_path_for_api, + parse_parent_specs, pick_random_distinct, resolve_item, sha256_hex, vote_touches_path, }; pub use rpc::handle_rpc_batch; diff --git a/server/src/api/rpc.rs b/server/src/api/rpc.rs index 5b91f5836625eedbb1cd9423168046e3fb576c17..5f7d50188f1381267402f2e57e671234ef5db2fd 100644 --- a/server/src/api/rpc.rs +++ b/server/src/api/rpc.rs @@ -8,6 +8,7 @@ use axum::{ Json, }; use rand::seq::SliceRandom; +use slug_types::paths::{ForumThreadUrl, GardenItemUrl, TildeOntologyPath}; use slug_types::*; use crate::{ @@ -27,9 +28,8 @@ use crate::{ use super::auth::verify_bearer_principal; use super::helpers::{ - compute_connectivity_stats, forum_thread_web_url, is_pair_voted, item_path_for_api, - item_path_for_api_in_room, now_ms, paginate_rankings, parse_parent_specs, pick_random_distinct, - resolve_item, vote_touches_path, + compute_connectivity_stats, is_pair_voted, now_ms, paginate_rankings, parse_parent_specs, + pick_random_distinct, resolve_item, vote_touches_path, }; use super::validate::{normalize_room_and_thread, validate_ingest_document}; @@ -184,7 +184,7 @@ fn compute_scope_rank_changes( }; if changed { changes.push(RankChange { - item: item_path_for_api_in_room(&item, room_wire), + item: GardenItemUrl::from_storage_str(&item, room_wire), before: b, after: a, }); @@ -206,7 +206,7 @@ fn compute_scope_rank_changes( parent: if parent.is_empty() { "/".to_string() } else { - item_path_for_api_in_room(parent, room_wire) + GardenItemUrl::from_storage_str(parent, room_wire).into_inner() }, changes, }) @@ -302,7 +302,7 @@ fn build_rank_response_for_content( .ranked .into_iter() .map(|r| RankRow { - item: item_path_for_api_in_room(r.item.as_str(), room_wire), + item: GardenItemUrl::from_stored(&r.item, room_wire), percent: if want_percent { Some((r.score / max_score) * 100.0) } else { @@ -315,10 +315,10 @@ fn build_rank_response_for_content( }) .collect(); - let prefixed_unranked: Vec = rankings + let prefixed_unranked: Vec = rankings .unranked_items .into_iter() - .map(|s| item_path_for_api_in_room(s.as_str(), room_wire)) + .map(|s| GardenItemUrl::from_stored(&s, room_wire)) .collect(); let (components, unranked_items) = if offset > 0 || limit.is_some() { @@ -537,13 +537,13 @@ async fn rpc_post( ( "npx slugsocial public garden pair".to_string(), "npx slugsocial public garden rank".to_string(), - forum_thread_web_url("public", &thread_id), + ForumThreadUrl::from_room_tag("public", &thread_id), ) } else { ( format!("npx slugsocial private {room_key} garden pair"), format!("npx slugsocial private {room_key} garden rank"), - forum_thread_web_url(&room_key, &thread_id), + ForumThreadUrl::from_room_tag(&room_key, &thread_id), ) }; @@ -664,7 +664,7 @@ async fn rpc_check( .ranked .into_iter() .map(|r| RankRow { - item: item_path_for_api_in_room(r.item.as_str(), &room_key), + item: GardenItemUrl::from_stored(&r.item, &room_key), score: r.score, percent: None, }) @@ -672,12 +672,12 @@ async fn rpc_check( }) .collect(); CheckScopeRanking { - parent: item_path_for_api_in_room(parent.as_str(), &room_key), + parent: GardenItemUrl::from_stored(parent, &room_key).into_inner(), components, unranked_items: scoped .unranked_items .into_iter() - .map(|it| item_path_for_api_in_room(it.as_str(), &room_key)) + .map(|it| GardenItemUrl::from_stored(&it, &room_key)) .collect(), } }) @@ -687,13 +687,13 @@ async fn rpc_check( vec![ "npx slugsocial public forum post --delegate ".to_string(), "npx slugsocial public forum list".to_string(), - forum_thread_web_url("public", &thread_id), + ForumThreadUrl::from_room_tag("public", &thread_id).into_inner(), ] } else { vec![ format!("npx slugsocial private {room_key} forum post --delegate "), format!("npx slugsocial private {room_key} forum list"), - forum_thread_web_url(&room_key, &thread_id), + ForumThreadUrl::from_room_tag(&room_key, &thread_id).into_inner(), ] }; @@ -717,7 +717,7 @@ fn rpc_list_forum_threads(reduced: &ReducerState, room: &str) -> ThreadsResponse .map(|((_, tag), ts)| ThreadSummary { thread: format!("#{tag}"), last_activity_ts: ts.last_activity_ts, - web: forum_thread_web_url(room, tag), + web: ForumThreadUrl::from_room_tag(room, tag), }) .collect(); out.sort_by(|a, b| b.last_activity_ts.cmp(&a.last_activity_ts)); @@ -885,7 +885,7 @@ fn rpc_search(reduced: &ReducerState, q: &str, limit: usize, principal: Option<& } if score > 0 { scored_items.push((score, SearchItemHit { - path: item_path_for_api(item.as_str()), + path: GardenItemUrl::from_storage_str(item.as_str(), "public"), body: content.item_bodies.get(item).map(|b| snippet_around(b, &words, 120)), })); } @@ -1058,8 +1058,8 @@ async fn rpc_get_pair(state: &AppState, room: String, parent_path: String) -> Re .collect(); let cs = compute_connectivity_stats(&content.ranking_group, &pool); Ok(RpcResult::Pair(PairResponse { - left: item_path_for_api_in_room(&left, &room), - right: item_path_for_api_in_room(&right, &room), + left: GardenItemUrl::from_storage_str(&left, &room), + right: GardenItemUrl::from_storage_str(&right, &room), left_body: lb, right_body: rb, threads: th, @@ -1137,7 +1137,7 @@ pub async fn handle_rpc_batch( if !content.items.contains(&item) { line_err( "item not found", - Some(format!("{} does not exist", item_path_for_api_in_room(&item_str, &room))), + Some(format!("{} does not exist", GardenItemUrl::from_storage_str(&item_str, &room))), ) } else { const MAX_ITEM_BODY: usize = 10_000; @@ -1160,7 +1160,7 @@ pub async fn handle_rpc_batch( .map(|s| s.iter().cloned().collect()) .unwrap_or_default(); line_ok(RpcResult::GardenItem(ItemResponse { - item: item_path_for_api_in_room(&item_str, &room), + item: GardenItemUrl::from_storage_str(&item_str, &room), body, truncated, body_len, @@ -1554,7 +1554,7 @@ pub async fn handle_rpc_batch( for r in items { let pct = want_percent.then(|| ((r.score - bot) / range * 100.0).clamp(0.0, 100.0)); ranked.push(RankRow { - item: item_path_for_api_in_room(r.item.as_str(), &room), + item: GardenItemUrl::from_storage_str(r.item.as_str(), &room), score: r.score, percent: pct, }); @@ -1574,7 +1574,7 @@ pub async fn handle_rpc_batch( let page: Vec = ranked .into_iter() .chain(unranked.into_iter().map(|it| RankRow { - item: item_path_for_api_in_room(&it, &room), + item: GardenItemUrl::from_storage_str(&it, &room), score: 0.0, percent: want_percent.then_some(0.0), })) @@ -1618,7 +1618,7 @@ pub async fn handle_rpc_batch( if !content.items.contains(&item) { line_err( "item not found", - Some(format!("{} does not exist", item_path_for_api_in_room(&item_str, &room))), + Some(format!("{} does not exist", GardenItemUrl::from_storage_str(&item_str, &room))), ) } else { let votes: Vec = content @@ -1629,8 +1629,8 @@ pub async fn handle_rpc_batch( .take(limit) .map(|v| VoteRow { ts: v.ts, - a: item_path_for_api_in_room(v.a.as_str(), &room), - b: item_path_for_api_in_room(v.b.as_str(), &room), + a: GardenItemUrl::from_stored(&v.a, &room), + b: GardenItemUrl::from_stored(&v.b, &room), ratio: format!("{}:{}", v.ratio_left, v.ratio_right), actor: Some(v.principal.clone()), body: v.body.clone(), @@ -1640,7 +1640,7 @@ pub async fn handle_rpc_batch( }) .unwrap_or_default(); line_ok(RpcResult::Matchup(MatchupResponse { - item: item_path_for_api_in_room(&item_str, &room), + item: GardenItemUrl::from_storage_str(&item_str, &room), votes, })) } @@ -1667,8 +1667,8 @@ pub async fn handle_rpc_batch( if a == item_str || b == item_str { Some(VoteRow { ts: e.ts, - a: item_path_for_api_in_room(&a, &room), - b: item_path_for_api_in_room(&b, &room), + a: GardenItemUrl::from_storage_str(&a, &room), + b: GardenItemUrl::from_storage_str(&b, &room), ratio: format!("{}:{}", ratio_left, ratio_right), actor: reduced.ingests_by_id.get(&e.post_id).map(|ing| ing.principal.clone()), body: explanation, @@ -1704,7 +1704,7 @@ pub async fn handle_rpc_batch( } }).collect(); line_ok(RpcResult::RankHistory(RankHistoryResponse { - item: item_path_for_api_in_room(&item_str, &room), + item: GardenItemUrl::from_storage_str(&item_str, &room), history, })) } @@ -1716,17 +1716,13 @@ pub async fn handle_rpc_batch( } else { let content = content_for_room(&reduced, &room); let parents: HashSet<&str> = content.item_children.keys().map(|s| s.as_str()).collect(); - let mut paths: Vec = content + let mut paths: Vec = content .items .iter() .filter(|p| !parents.contains(p.as_str())) - .map(|p| p.as_str().to_string()) - .collect(); - paths.sort(); - let paths: Vec = paths - .into_iter() - .map(|p| item_path_for_api_in_room(&p, &room)) + .map(|p| GardenItemUrl::from_stored(p, &room)) .collect(); + paths.sort_by(|a, b| a.as_str().cmp(b.as_str())); line_ok(RpcResult::Leaves(LeavesResponse { paths })) } }, @@ -1743,24 +1739,13 @@ pub async fn handle_rpc_batch( let mut v: Vec = roots.iter() .map(|path| { let children = content.item_children.get(path.as_str()).map(|s| s.len()).unwrap_or(0); - let path_label = CanonicalItemUrl::parse(path.as_str()) - .and_then(|c| { - c.tilde_tail().map(|t| { - if t.is_empty() { - "~/".to_string() - } else { - format!("~/{}", t) - } - }) - }) - .unwrap_or_else(|| path.to_string()); PathSummary { - path: path_label, + path: TildeOntologyPath::from_stored(path), children, - web: item_path_for_api_in_room(path.as_str(), &room), + web: GardenItemUrl::from_stored(path, &room), } }).collect(); - v.sort_by(|a, b| a.path.cmp(&b.path)); + v.sort_by(|a, b| a.path.as_str().cmp(b.path.as_str())); v }) .unwrap_or_default(); @@ -1790,8 +1775,8 @@ pub async fn handle_rpc_batch( .take(limit) .map(|v| VoteRow { ts: v.ts, - a: item_path_for_api_in_room(v.a.as_str(), &room), - b: item_path_for_api_in_room(v.b.as_str(), &room), + a: GardenItemUrl::from_stored(&v.a, &room), + b: GardenItemUrl::from_stored(&v.b, &room), ratio: format!("{}:{}", v.ratio_left, v.ratio_right), actor: Some(v.principal.clone()), body: v.body.clone(), diff --git a/server/src/api/validate.rs b/server/src/api/validate.rs index 27150577982f52cbd6d11bb93654dc3d4cf75cc5..a51c783ee9785569b5a44c0b1572471fe00d174b 100644 --- a/server/src/api/validate.rs +++ b/server/src/api/validate.rs @@ -7,8 +7,9 @@ use crate::{ path_types::CanonicalItemUrl, reducer::{ReducerState, ScopeId}, }; +use slug_types::paths::GardenItemUrl; -use super::helpers::{item_path_for_api, resolve_item}; +use super::helpers::resolve_item; #[derive(Debug)] pub struct ValidatedIngest { @@ -22,6 +23,10 @@ pub fn validate_ingest_document( text: &str, scope: &ScopeId, ) -> Result)> { + let room_wire = match scope { + ScopeId::Public => "public", + ScopeId::Room(r) => r.as_str(), + }; let public_content = reduced.public(); let scoped_content = match scope { ScopeId::Public => None, @@ -61,14 +66,14 @@ pub fn validate_ingest_document( let Some(body_text) = body else { return Err(( StatusCode::BAD_REQUEST, - format!("item missing body: {}", item_path_for_api(&item)), + format!("item missing body: {}", GardenItemUrl::from_storage_str(&item, room_wire)), Some("items must be declared with bodies, e.g. `~/path/item { ... }`".to_string()), )); }; if body_text.trim().is_empty() { return Err(( StatusCode::BAD_REQUEST, - format!("item body is empty: {}", item_path_for_api(&item)), + format!("item body is empty: {}", GardenItemUrl::from_storage_str(&item, room_wire)), Some("write at least one sentence inside `{ ... }`".to_string()), )); } @@ -101,7 +106,7 @@ pub fn validate_ingest_document( let key = CanonicalItemUrl((*it).clone()); !defined_in_doc.contains(*it) && !item_exists(&key) }) - .map(|it| item_path_for_api(it)) + .map(|it| GardenItemUrl::from_storage_str(it, room_wire).into_inner()) .collect(); if !missing.is_empty() { return Err(( @@ -119,7 +124,7 @@ pub fn validate_ingest_document( let key = CanonicalItemUrl((*it).clone()); !defined_in_doc.contains(*it) && !body_exists(&key) }) - .map(|it| item_path_for_api(it)) + .map(|it| GardenItemUrl::from_storage_str(it, room_wire).into_inner()) .collect(); if !missing_body.is_empty() { return Err(( diff --git a/server/src/canonical_path.rs b/server/src/canonical_path.rs index 5c0febe883d8b3978a25896ed0d71df8509a8717..8a1998025121838b0867f8c5ddbd28aea41a23d2 100644 --- a/server/src/canonical_path.rs +++ b/server/src/canonical_path.rs @@ -1,92 +1,3 @@ -//! Normalization for thread tags and ontology item URLs (DSL ↔ stored canonical form). -//! Not event types — see `events` and `path_types`. +//! Re-exports — implementations live in `slug-types` (`paths` module). -/// Thread / public tag: stored without leading `#`, lowercase. -pub fn canonicalize_tag(input: &str) -> String { - input.trim().trim_start_matches('#').to_lowercase() -} - -/// Ontology item reference → canonical absolute URL on the slug host. -pub fn canonicalize_item(input: &str) -> String { - let s = input.trim(); - if s.is_empty() { - return String::new(); - } - - if let Some(rest) = s.strip_prefix("https://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let host = host.trim().to_lowercase(); - if tail.is_empty() { - return format!("https://{}", host); - } else { - return format!("https://{}/{}", host, tail); - } - } - if let Some(rest) = s.strip_prefix("http://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let host = host.trim().to_lowercase(); - if tail.is_empty() { - return format!("http://{}", host); - } else { - return format!("http://{}/{}", host, tail); - } - } - - let is_tilde = s.starts_with("~/"); - let rest = s.strip_prefix("~/").or_else(|| s.strip_prefix("/")).unwrap_or(s); - - let tail = rest - .split('/') - .filter_map(|seg| { - let t = seg.trim(); - if t.is_empty() { - None - } else { - Some(t.to_lowercase()) - } - }) - .collect::>() - .join("/"); - - if is_tilde { - format!("https://slug.social/~/{}", tail) - } else if tail.is_empty() { - "https://slug.social".to_string() - } else { - format!("https://slug.social/{}", tail) - } -} - -pub fn item_path_segments(input: &str) -> Vec { - let canonical = canonicalize_item(input); - if canonical.is_empty() { - return vec![]; - } - - if let Some(rest) = canonical.strip_prefix("https://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let mut out = vec![format!("https://{}", host)]; - out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); - return out; - } - if let Some(rest) = canonical.strip_prefix("http://") { - let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); - let mut out = vec![format!("http://{}", host)]; - out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); - return out; - } - - canonical - .split('/') - .filter(|s| !s.is_empty()) - .map(|s| s.to_string()) - .collect() -} - -pub fn item_parent_path(input: &str) -> Option { - let segs = item_path_segments(input); - if segs.len() <= 1 { - return None; - } - Some(segs[..segs.len() - 1].join("/")) -} +pub use slug_types::paths::{canonicalize_item, canonicalize_tag, item_parent_path, item_path_segments}; diff --git a/server/src/html/forum.rs b/server/src/html/forum.rs index ab7970feca47a4431ddc19dab2c8186180b78413..f789e347dc136f8ffd356f4492f0bd76cea04bf0 100644 --- a/server/src/html/forum.rs +++ b/server/src/html/forum.rs @@ -697,7 +697,7 @@ async fn thread_view_inner( let offset = q.offset.unwrap_or(0); let page_ids: Vec = all_ids.into_iter().skip(offset).take(PAGE_SIZE).collect(); - let (display_ingests, subtitle) = { + let (display_ingests, _subtitle) = { let reduced = state.reduced.read().await; let ingests = page_ids .iter() diff --git a/server/src/path_types.rs b/server/src/path_types.rs index cad58cac7da948f92d6ea2a5587d917ab21d57ea..4c8075bbd488f1b9a5eda9e03ced83f6238c6d5e 100644 --- a/server/src/path_types.rs +++ b/server/src/path_types.rs @@ -1,260 +1,3 @@ -//! Path representation types. -//! -//! The codebase currently treats item identifiers as strings in a few different -//! encodings: -//! - user/DSL input like `~/a/b` -//! - canonical item URLs like `https://slug.social/~/a/b` -//! - relative paths within a rooted tree view (e.g. `llms/openai` under a root) -//! -//! This module adds lightweight newtypes so code can be explicit about what it -//! expects without changing core storage formats. -//! -//! **Storage vs wire:** [`CanonicalItemUrl`] values are shared across scopes -//! (`https://slug.social/~/…`); which [`crate::reducer::ContentState`] they live in -//! is determined by scope, not by embedding the room id in the string. For JSON/RPC -//! and browser links in a private room, use [`crate::api::helpers::item_path_for_api_in_room`] -//! so ontology items become `https://slug.social/r/{short}/{slug}/~/…`. +//! Re-exports — implementations live in `slug-types` (`paths` module). -use std::borrow::Borrow; -use std::fmt; - -use serde::{Deserialize, Serialize}; - -use crate::canonical_path::canonicalize_item; - -/// Canonical item identifier as produced by `canonical_path::canonicalize_item`. -/// -/// In practice this is usually: -/// - `https://slug.social/~/...` for ontology items, or -/// - `https://...` / `http://...` for URL items. -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -pub struct CanonicalItemUrl(pub String); - -impl CanonicalItemUrl { - pub fn parse(input: &str) -> Option { - let c = canonicalize_item(input); - if c.is_empty() { - None - } else { - Some(Self(c)) - } - } - - pub fn as_str(&self) -> &str { - &self.0 - } - - /// Returns the `~/...` tail for ontology items (`https://slug.social/~/...`). - pub fn tilde_tail(&self) -> Option<&str> { - self.0.strip_prefix("https://slug.social/~/") - } - - /// Returns the final non-empty `/`-separated segment of the path. - /// - /// `https://slug.social/~/a/b/c` → `"c"` - /// `https://slug.social/~/a` → `"a"` - pub fn last_segment(&self) -> &str { - self.0 - .rsplit('/') - .find(|s| !s.is_empty()) - .unwrap_or(self.0.as_str()) - } - - /// The ontology root key as stored in `item_children`: `"https://slug.social/~"`. - /// Use this (not `parse("~/")`) when looking up top-level children. - pub fn ontology_root() -> Self { - Self("https://slug.social/~".to_string()) - } - - /// Returns the parent of this canonical item URL by stripping the last - /// path segment, or `None` if there is no parent (already at root). - /// - /// `https://slug.social/~/a/b/c` → `Some("https://slug.social/~/a/b")` - /// `https://slug.social/~/a` → `Some("https://slug.social/~")` - /// `https://slug.social/~/` → `None` (tilde root) - pub fn parent(&self) -> Option { - // tilde_tail() is None for non-ontology URLs and "" for the root ~/ - if self.tilde_tail().map(|t| t.is_empty()).unwrap_or(true) { - return None; - } - // Strip everything from the last '/' onwards. - let last_slash = self.0.rfind('/')?; - let parent_str = &self.0[..last_slash]; - if parent_str.is_empty() { - None - } else { - Some(Self(parent_str.to_string())) - } - } - - /// Segments of an ontology path suitable for breadcrumb rendering. - /// Strips the `https://slug.social` prefix and returns the `~/…` parts. - /// - /// `https://slug.social/~/a/b` → `["~", "a", "b"]` - /// `https://slug.social/~/` → `["~"]` - pub fn tilde_segments(&self) -> Vec<&str> { - match self.tilde_tail() { - Some(tail) if !tail.is_empty() => { - std::iter::once("~") - .chain(tail.split('/').filter(|s| !s.is_empty())) - .collect() - } - Some(_) => vec!["~"], - None => vec![], - } - } -} - -impl fmt::Display for CanonicalItemUrl { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - self.0.fmt(f) - } -} - -/// Allow `HashMap` to be searched by `&str`. -impl Borrow for CanonicalItemUrl { - fn borrow(&self) -> &str { - &self.0 - } -} - -impl PartialEq for CanonicalItemUrl { - fn eq(&self, other: &str) -> bool { - self.0 == other - } -} - -impl PartialEq<&str> for CanonicalItemUrl { - fn eq(&self, other: &&str) -> bool { - self.0 == *other - } -} - -impl PartialEq for CanonicalItemUrl { - fn eq(&self, other: &String) -> bool { - &self.0 == other - } -} - -/// A `~/...` input path (as used in the DSL and UX). -/// -/// This is not canonicalized; it is a presentation/input form. -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -pub struct TildePath(pub String); - -impl TildePath { - pub fn new(input: &str) -> Option { - let s = input.trim(); - if s.starts_with("~/") && s.len() > 2 { - Some(Self(s.to_string())) - } else if s == "~/" { - Some(Self("~/".to_string())) - } else { - None - } - } - - pub fn as_str(&self) -> &str { - &self.0 - } - - pub fn canonicalize(&self) -> Option { - CanonicalItemUrl::parse(&self.0) - } -} - -impl fmt::Display for TildePath { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - self.0.fmt(f) - } -} - -/// A path relative to a chosen root in a tree UI. -/// -/// This is intended for compact state encodings (blobs). It must be joined to a -/// root `CanonicalItemUrl` (typically an ontology root) to become a full item. -#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] -pub struct RelativePath(pub String); - -impl RelativePath { - pub fn new(input: &str) -> Option { - let s = input.trim().trim_matches('/'); - if s.is_empty() { - Some(Self(String::new())) - } else { - // Keep this permissive: the DSL parser is the main gatekeeper. - Some(Self(s.to_string())) - } - } - - pub fn as_str(&self) -> &str { - &self.0 - } - - /// Join this relative path under a canonical ontology root - /// (`https://slug.social/~/...`) to form a canonical item URL. - pub fn join_under_ontology_root(&self, root: &CanonicalItemUrl) -> Option { - let base = root.tilde_tail()?; - // base is the tail after https://slug.social/~/, e.g. "models" or "models/llms" - let joined = if base.is_empty() { - if self.0.is_empty() { - "~/".to_string() - } else { - format!("~/{}", self.0) - } - } else if self.0.is_empty() { - format!("~/{}", base) - } else { - format!("~/{}/{}", base.trim_end_matches('/'), self.0) - }; - CanonicalItemUrl::parse(&joined) - } -} - -impl fmt::Display for RelativePath { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - self.0.fmt(f) - } -} - - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn canonical_parent_deep() { - let c = CanonicalItemUrl::parse("~/a/b/c").unwrap(); - assert_eq!(c.parent().unwrap().as_str(), "https://slug.social/~/a/b"); - } - - #[test] - fn canonical_parent_one_level() { - let c = CanonicalItemUrl::parse("~/a").unwrap(); - assert_eq!(c.parent().unwrap().as_str(), "https://slug.social/~"); - } - - #[test] - fn canonical_parent_root_is_none() { - let root = CanonicalItemUrl::parse("~/").unwrap(); - assert!(root.parent().is_none()); - } - - #[test] - fn tilde_segments_deep() { - let c = CanonicalItemUrl::parse("~/a/b").unwrap(); - assert_eq!(c.tilde_segments(), vec!["~", "a", "b"]); - } - - #[test] - fn tilde_segments_root() { - let c = CanonicalItemUrl::parse("~/").unwrap(); - assert_eq!(c.tilde_segments(), vec!["~"]); - } - - #[test] - fn tilde_segments_non_ontology_is_empty() { - let c = CanonicalItemUrl::parse("https://example.com/foo").unwrap(); - assert_eq!(c.tilde_segments(), Vec::<&str>::new()); - } -} +pub use slug_types::paths::{CanonicalItemUrl, RelativePath, TildePath}; diff --git a/types/src/lib.rs b/types/src/lib.rs index c1cde3b783d03b02783385b5f07659fb112aae3e..5fc867bf2af84c2ca63fa1cdd03413110ad784a3 100644 --- a/types/src/lib.rs +++ b/types/src/lib.rs @@ -1,7 +1,13 @@ use serde::{Deserialize, Serialize}; +pub mod paths; pub mod timeago; +pub use paths::{ + canonicalize_item, canonicalize_tag, item_parent_path, item_path_segments, CanonicalItemUrl, + ForumThreadUrl, GardenItemUrl, RelativePath, TildeOntologyPath, TildePath, +}; + #[derive(Debug, Serialize, Deserialize)] pub struct ApiError { pub ok: bool, @@ -12,7 +18,7 @@ pub struct ApiError { #[derive(Debug, Clone, Serialize, Deserialize)] pub struct RankRow { - pub item: String, + pub item: GardenItemUrl, pub score: f64, /// Normalized score as a percentage of the top item (0–100). Present when ?percent=true. #[serde(skip_serializing_if = "Option::is_none")] @@ -43,7 +49,7 @@ pub struct RankComponent { #[derive(Debug, Serialize, Deserialize)] pub struct RankResponse { pub components: Vec, - pub unranked_items: Vec, + pub unranked_items: Vec, } /// Graph connectivity stats for a scope, returned with pair suggestions. @@ -63,8 +69,8 @@ pub struct ConnectivityStats { #[derive(Debug, Serialize, Deserialize)] pub struct PairResponse { - pub left: String, - pub right: String, + pub left: GardenItemUrl, + pub right: GardenItemUrl, pub left_body: Option, pub right_body: Option, /// Thread tags that discuss either item (connective tissue to forum). @@ -79,7 +85,7 @@ pub struct PairResponse { pub struct NextMoves { pub pair: String, pub rank: String, - pub web: String, + pub web: ForumThreadUrl, } #[derive(Debug, Serialize, Deserialize)] @@ -90,14 +96,14 @@ pub struct PathsResponse { /// Leaf items only (no children). For search / "full path list" — does not scale, works for now. #[derive(Debug, Serialize, Deserialize)] pub struct LeavesResponse { - pub paths: Vec, + pub paths: Vec, } #[derive(Debug, Serialize, Deserialize)] pub struct PathSummary { - pub path: String, + pub path: TildeOntologyPath, pub children: usize, - pub web: String, + pub web: GardenItemUrl, } #[derive(Debug, Clone, Serialize, Deserialize)] @@ -109,7 +115,7 @@ pub struct ThreadsResponse { pub struct ThreadSummary { pub thread: String, pub last_activity_ts: i64, - pub web: String, + pub web: ForumThreadUrl, } #[derive(Debug, Serialize, Deserialize)] @@ -178,7 +184,7 @@ pub struct IngestRow { #[derive(Debug, Serialize, Deserialize)] pub struct ItemResponse { - pub item: String, + pub item: GardenItemUrl, pub body: Option, /// True when the body was truncated due to size. Fetch with `?full=true` for the complete body. #[serde(default, skip_serializing_if = "std::ops::Not::not")] @@ -201,15 +207,15 @@ pub struct RecentVotesResponse { /// Vote history for one item (matchup: wins/losses + thread per vote). #[derive(Debug, Serialize, Deserialize)] pub struct MatchupResponse { - pub item: String, + pub item: GardenItemUrl, pub votes: Vec, } #[derive(Debug, Clone, Serialize, Deserialize)] pub struct VoteRow { pub ts: i64, - pub a: String, - pub b: String, + pub a: GardenItemUrl, + pub b: GardenItemUrl, pub ratio: String, /// Principal username when present (stored form, no `@`). pub actor: Option, @@ -510,7 +516,7 @@ pub struct RankPosition { /// How one item's rank changed after a vote. #[derive(Debug, Clone, Serialize, Deserialize)] pub struct RankChange { - pub item: String, + pub item: GardenItemUrl, /// Position before the vote. None = was unranked (no voted connections in this scope). pub before: Option, /// Position after the vote. None = became unranked (e.g. component split, unlikely). @@ -544,7 +550,7 @@ pub struct CheckScopeRanking { /// Parent scope path (e.g. "/models" or "/" for root). pub parent: String, pub components: Vec, - pub unranked_items: Vec, + pub unranked_items: Vec, } #[derive(Debug, Serialize, Deserialize)] @@ -567,7 +573,7 @@ pub struct SearchResponse { #[derive(Debug, Serialize, Deserialize)] pub struct SearchItemHit { - pub path: String, + pub path: GardenItemUrl, #[serde(skip_serializing_if = "Option::is_none")] pub body: Option, } @@ -614,7 +620,7 @@ pub struct RankHistoryRow { #[derive(Debug, Serialize, Deserialize)] pub struct RankHistoryResponse { - pub item: String, + pub item: GardenItemUrl, pub history: Vec, } diff --git a/types/src/paths.rs b/types/src/paths.rs new file mode 100644 index 0000000000000000000000000000000000000000..2950a7502255927583fbacdfd2adb700f0b0c221 --- /dev/null +++ b/types/src/paths.rs @@ -0,0 +1,498 @@ +//! Canonical paths, storage ids, and JSON href newtypes. All normalization and +//! room-aware URL rules for items live here. + +use std::borrow::Borrow; +use std::fmt; + +use serde::{Deserialize, Serialize}; + +// --------------------------------------------------------------------------- +// Normalization (moved from server `canonical_path`) +// --------------------------------------------------------------------------- + +/// Thread / public tag: stored without leading `#`, lowercase. +pub fn canonicalize_tag(input: &str) -> String { + input.trim().trim_start_matches('#').to_lowercase() +} + +/// Ontology item reference → canonical absolute URL on the slug host. +pub fn canonicalize_item(input: &str) -> String { + let s = input.trim(); + if s.is_empty() { + return String::new(); + } + + if let Some(rest) = s.strip_prefix("https://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let host = host.trim().to_lowercase(); + if tail.is_empty() { + return format!("https://{}", host); + } else { + return format!("https://{}/{}", host, tail); + } + } + if let Some(rest) = s.strip_prefix("http://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let host = host.trim().to_lowercase(); + if tail.is_empty() { + return format!("http://{}", host); + } else { + return format!("http://{}/{}", host, tail); + } + } + + let is_tilde = s.starts_with("~/"); + let rest = s.strip_prefix("~/").or_else(|| s.strip_prefix("/")).unwrap_or(s); + + let tail = rest + .split('/') + .filter_map(|seg| { + let t = seg.trim(); + if t.is_empty() { + None + } else { + Some(t.to_lowercase()) + } + }) + .collect::>() + .join("/"); + + if is_tilde { + format!("https://slug.social/~/{}", tail) + } else if tail.is_empty() { + "https://slug.social".to_string() + } else { + format!("https://slug.social/{}", tail) + } +} + +pub fn item_path_segments(input: &str) -> Vec { + let canonical = canonicalize_item(input); + if canonical.is_empty() { + return vec![]; + } + + if let Some(rest) = canonical.strip_prefix("https://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let mut out = vec![format!("https://{}", host)]; + out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); + return out; + } + if let Some(rest) = canonical.strip_prefix("http://") { + let (host, tail) = rest.split_once('/').map_or((rest, ""), |(h, t)| (h, t)); + let mut out = vec![format!("http://{}", host)]; + out.extend(tail.split('/').filter(|s| !s.is_empty()).map(|s| s.to_string())); + return out; + } + + canonical + .split('/') + .filter(|s| !s.is_empty()) + .map(|s| s.to_string()) + .collect() +} + +pub fn item_parent_path(input: &str) -> Option { + let segs = item_path_segments(input); + if segs.len() <= 1 { + return None; + } + Some(segs[..segs.len() - 1].join("/")) +} + +// --------------------------------------------------------------------------- +// Storage + input path newtypes +// --------------------------------------------------------------------------- + +/// Canonical item identifier as produced by [`canonicalize_item`]. +/// +/// Shared across all scopes; room is not embedded. Usually +/// `https://slug.social/~/…` or an external `http(s)://…` URL item. +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +pub struct CanonicalItemUrl(pub String); + +impl CanonicalItemUrl { + pub fn parse(input: &str) -> Option { + let c = canonicalize_item(input); + if c.is_empty() { + None + } else { + Some(Self(c)) + } + } + + pub fn as_str(&self) -> &str { + &self.0 + } + + pub fn tilde_tail(&self) -> Option<&str> { + self.0.strip_prefix("https://slug.social/~/") + } + + pub fn last_segment(&self) -> &str { + self.0 + .rsplit('/') + .find(|s| !s.is_empty()) + .unwrap_or(self.0.as_str()) + } + + pub fn ontology_root() -> Self { + Self("https://slug.social/~".to_string()) + } + + pub fn parent(&self) -> Option { + if self.tilde_tail().map(|t| t.is_empty()).unwrap_or(true) { + return None; + } + let last_slash = self.0.rfind('/')?; + let parent_str = &self.0[..last_slash]; + if parent_str.is_empty() { + None + } else { + Some(Self(parent_str.to_string())) + } + } + + pub fn tilde_segments(&self) -> Vec<&str> { + match self.tilde_tail() { + Some(tail) if !tail.is_empty() => { + std::iter::once("~") + .chain(tail.split('/').filter(|s| !s.is_empty())) + .collect() + } + Some(_) => vec!["~"], + None => vec![], + } + } + + /// `~/…` list label for ontology items (paths index, CLI). + pub fn tilde_list_label(&self) -> TildeOntologyPath { + TildeOntologyPath::from_stored(self) + } + + /// Absolute href for JSON/RPC and browsers for this stored id in `room`. + pub fn json_href(&self, room_wire: &str) -> GardenItemUrl { + GardenItemUrl::from_stored(self, room_wire) + } +} + +impl fmt::Display for CanonicalItemUrl { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +impl Borrow for CanonicalItemUrl { + fn borrow(&self) -> &str { + &self.0 + } +} + +impl PartialEq for CanonicalItemUrl { + fn eq(&self, other: &str) -> bool { + self.0 == other + } +} + +impl PartialEq<&str> for CanonicalItemUrl { + fn eq(&self, other: &&str) -> bool { + self.0 == *other + } +} + +impl PartialEq for CanonicalItemUrl { + fn eq(&self, other: &String) -> bool { + &self.0 == other + } +} + +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +pub struct TildePath(pub String); + +impl TildePath { + pub fn new(input: &str) -> Option { + let s = input.trim(); + if s.starts_with("~/") && s.len() > 2 { + Some(Self(s.to_string())) + } else if s == "~/" { + Some(Self("~/".to_string())) + } else { + None + } + } + + pub fn as_str(&self) -> &str { + &self.0 + } + + pub fn canonicalize(&self) -> Option { + CanonicalItemUrl::parse(&self.0) + } +} + +impl fmt::Display for TildePath { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +pub struct RelativePath(pub String); + +impl RelativePath { + pub fn new(input: &str) -> Option { + let s = input.trim().trim_matches('/'); + if s.is_empty() { + Some(Self(String::new())) + } else { + Some(Self(s.to_string())) + } + } + + pub fn as_str(&self) -> &str { + &self.0 + } + + pub fn join_under_ontology_root(&self, root: &CanonicalItemUrl) -> Option { + let base = root.tilde_tail()?; + let joined = if base.is_empty() { + if self.0.is_empty() { + "~/".to_string() + } else { + format!("~/{}", self.0) + } + } else if self.0.is_empty() { + format!("~/{}", base) + } else { + format!("~/{}/{}", base.trim_end_matches('/'), self.0) + }; + CanonicalItemUrl::parse(&joined) + } +} + +impl fmt::Display for RelativePath { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +// --------------------------------------------------------------------------- +// Wire / JSON: correct-by-construction hrefs +// --------------------------------------------------------------------------- + +fn api_path_or_url(item: &str) -> String { + if item.starts_with("http://") || item.starts_with("https://") { + item.to_string() + } else { + format!("/{}", item) + } +} + +/// Ontology item as serialized in JSON (absolute URL or `/`-prefixed path). +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(transparent)] +pub struct GardenItemUrl(pub String); + +impl GardenItemUrl { + pub fn as_str(&self) -> &str { + &self.0 + } + + pub fn into_inner(self) -> String { + self.0 + } + + /// Stored canonical id + RPC `room` field (`"public"` or `"short/slug"`). + pub fn from_stored(stored: &CanonicalItemUrl, room_wire: &str) -> Self { + Self(garden_href_string(stored.as_str(), room_wire)) + } + + /// Like [`Self::from_stored`] but accepts a string that may already be canonical. + pub fn from_storage_str(stored: &str, room_wire: &str) -> Self { + Self(garden_href_string(stored, room_wire)) + } +} + +impl fmt::Display for GardenItemUrl { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +fn garden_href_string(item: &str, room_wire: &str) -> String { + let room = room_wire.trim(); + if room.is_empty() || room == "public" { + return api_path_or_url(item); + } + let Some((short, slug)) = room.split_once('/') else { + return api_path_or_url(item); + }; + if short.is_empty() || slug.is_empty() { + return api_path_or_url(item); + } + let Some(c) = CanonicalItemUrl::parse(item) else { + return api_path_or_url(item); + }; + let root = CanonicalItemUrl::ontology_root(); + let item_norm = c.as_str().trim_end_matches('/'); + let root_norm = root.as_str().trim_end_matches('/'); + if let Some(tail) = c.tilde_tail() { + return if tail.is_empty() { + format!("https://slug.social/r/{short}/{slug}/~") + } else { + format!("https://slug.social/r/{short}/{slug}/~/{}", tail) + }; + } + if item_norm == root_norm { + return format!("https://slug.social/r/{short}/{slug}/~"); + } + api_path_or_url(item) +} + +/// Forum thread URL for JSON (`/t/…` or `/r/…/t/…` on slug.social). +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(transparent)] +pub struct ForumThreadUrl(pub String); + +impl ForumThreadUrl { + pub fn as_str(&self) -> &str { + &self.0 + } + + pub fn into_inner(self) -> String { + self.0 + } + + pub fn from_room_tag(room_wire: &str, thread_tag: &str) -> Self { + let room = room_wire.trim(); + let tag = thread_tag.trim().trim_start_matches('#'); + Self(if room.is_empty() || room == "public" { + format!("https://slug.social/t/{tag}") + } else if let Some((short, slug)) = room.split_once('/') { + if short.is_empty() || slug.is_empty() { + format!("https://slug.social/t/{tag}") + } else { + format!("https://slug.social/r/{short}/{slug}/t/{tag}") + } + } else { + format!("https://slug.social/t/{tag}") + }) + } +} + +impl fmt::Display for ForumThreadUrl { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +/// `~/a/b` style path for list UIs (paths index `path` field). +#[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)] +#[serde(transparent)] +pub struct TildeOntologyPath(pub String); + +impl TildeOntologyPath { + pub fn from_stored(c: &CanonicalItemUrl) -> Self { + let s = match c.tilde_tail() { + Some(tail) if !tail.is_empty() => format!("~/{}", tail), + Some(_) => "~/".to_string(), + None => c.to_string(), + }; + Self(s) + } + + pub fn as_str(&self) -> &str { + &self.0 + } +} + +impl fmt::Display for TildeOntologyPath { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.0.fmt(f) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn canonical_parent_deep() { + let c = CanonicalItemUrl::parse("~/a/b/c").unwrap(); + assert_eq!(c.parent().unwrap().as_str(), "https://slug.social/~/a/b"); + } + + #[test] + fn canonical_parent_one_level() { + let c = CanonicalItemUrl::parse("~/a").unwrap(); + assert_eq!(c.parent().unwrap().as_str(), "https://slug.social/~"); + } + + #[test] + fn canonical_parent_root_is_none() { + let root = CanonicalItemUrl::parse("~/").unwrap(); + assert!(root.parent().is_none()); + } + + #[test] + fn tilde_segments_deep() { + let c = CanonicalItemUrl::parse("~/a/b").unwrap(); + assert_eq!(c.tilde_segments(), vec!["~", "a", "b"]); + } + + #[test] + fn tilde_segments_root() { + let c = CanonicalItemUrl::parse("~/").unwrap(); + assert_eq!(c.tilde_segments(), vec!["~"]); + } + + #[test] + fn tilde_segments_non_ontology_is_empty() { + let c = CanonicalItemUrl::parse("https://example.com/foo").unwrap(); + assert_eq!(c.tilde_segments(), Vec::<&str>::new()); + } + + #[test] + fn garden_public_passthrough_https() { + let u = "https://slug.social/~/a/b"; + assert_eq!(GardenItemUrl::from_storage_str(u, "public").as_str(), u); + } + + #[test] + fn garden_private_room_prefixes_ontology() { + assert_eq!( + GardenItemUrl::from_storage_str("https://slug.social/~/topic/x", "9ab12cd/my-room").as_str(), + "https://slug.social/r/9ab12cd/my-room/~/topic/x" + ); + } + + #[test] + fn garden_private_room_ontology_root() { + assert_eq!( + GardenItemUrl::from_storage_str("https://slug.social/~", "9ab12cd/my-room").as_str(), + "https://slug.social/r/9ab12cd/my-room/~" + ); + assert_eq!( + GardenItemUrl::from_storage_str("https://slug.social/~/", "9ab12cd/my-room").as_str(), + "https://slug.social/r/9ab12cd/my-room/~" + ); + } + + #[test] + fn garden_external_url_untouched_in_private_room() { + let u = "https://example.com/z"; + assert_eq!(GardenItemUrl::from_storage_str(u, "9ab12cd/my-room").as_str(), u); + } + + #[test] + fn forum_web_public_vs_room() { + assert_eq!( + ForumThreadUrl::from_room_tag("public", "debate").as_str(), + "https://slug.social/t/debate" + ); + assert_eq!( + ForumThreadUrl::from_room_tag("9ab12cd/my-room", "#debate").as_str(), + "https://slug.social/r/9ab12cd/my-room/t/debate" + ); + } +} Side B — contributor: tommy-mor Side B — commit message: [6d04afc2] refactor Side B — unified diff (full patch): diff --git a/Cargo.lock b/Cargo.lock index 2cea973082716e761ef6f5dd5886acc08ff9aac0..8c43fb75c472b102e6e1d3b837dce3355be898f2 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -17,6 +17,28 @@ version = "1.0.102" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7f202df86484c868dbad7eaa557ef785d5c66295e41b460ef922eca0723b842c" +[[package]] +name = "async-stream" +version = "0.3.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b5a71a6f37880a80d1d7f19efd781e4b5de42c88f0722cc13bcb6cc2cfe8476" +dependencies = [ + "async-stream-impl", + "futures-core", + "pin-project-lite", +] + +[[package]] +name = "async-stream-impl" +version = "0.3.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c7c24de15d275a1ecfd47a380fb4d5ec9bfe0933f309ed5e705b775596a3574d" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + [[package]] name = "async-trait" version = "0.1.89" @@ -1242,9 +1264,11 @@ dependencies = [ name = "sorter2-server" version = "0.0.1" dependencies = [ + "async-stream", "axum", "axum-extra", "dotenvy", + "futures-util", "maud", "reqwest", "serde", diff --git a/server/Cargo.toml b/server/Cargo.toml index bd600138b613bd0f546bdec217a5334cdcb20aa5..c940acb687fb141d21760a3d6656172013cf6f41 100644 --- a/server/Cargo.toml +++ b/server/Cargo.toml @@ -18,6 +18,8 @@ tracing = "0.1" tracing-subscriber = { version = "0.3", features = ["env-filter"] } reqwest = { version = "0.12", features = ["json"] } dotenvy = "0.15" +async-stream = "0.3" +futures-util = { version = "0.3", default-features = false, features = ["std"] } [dev-dependencies] reqwest = { version = "0.12", features = ["json"] } diff --git a/server/src/api/ui_html.rs b/server/src/api/ui_html.rs index b33a84e8bb5e817b26592868d88090e6d664d950..7af6527d03c483f33f3469ce6766c01a554c5fe3 100644 --- a/server/src/api/ui_html.rs +++ b/server/src/api/ui_html.rs @@ -6,7 +6,8 @@ use axum::{ use std::collections::HashMap; use crate::{ - html::{entity_section, input_panel, js_string_literal, ranking_panel, JsBuilder}, + fetch, + html::{input_panel, js_string_literal, ranking_panel, JsBuilder}, parser::parse_reddit_url, path_types::ItemId, reddit::ensure_partial_tree, @@ -89,18 +90,8 @@ pub async fn post_ui_html( }, HtmlUiAction::FetchEntity { item } => { let id = parse_item_param(&item); - if id.is_root() { - return ui_js_warn("nothing to fetch for the root").into_response(); - } - state.queue_entity_fetch(id.clone()); - let tree = state.tree.read().await; - let empty = crate::reducer::NodeState::default(); - let node = tree.get(&id).unwrap_or(&empty); - let panel = entity_section(&id, node, true); - JsBuilder::new() - .morph_selector("#entity-section", panel) - .into_response() - }, + fetch::fetch_entity_stream(state, id).into_response() + } } } diff --git a/server/src/fetch/html.rs b/server/src/fetch/html.rs new file mode 100644 index 0000000000000000000000000000000000000000..63634508496e224c38b9ec0308b7a6086462f925 --- /dev/null +++ b/server/src/fetch/html.rs @@ -0,0 +1,67 @@ +//! Markup for entity import / “Fetch from Reddit” (`POST /ui`, SSE response). + +use maud::{html, Markup}; + +use crate::{ + form_template::template_json_compact, + path_types::ItemId, + reddit::is_fetchable, + reducer::NodeState, + ui_action::UI_RPC_FIELD, +}; + +fn entity_panel(node: &NodeState) -> Markup { + html! { + @if let Some(data) = &node.data { + div id="entity-panel" class="entity-card" { + h2 { (data.title) } + @if let Some(author) = &data.author { + p class="muted small" { "by " (author) } + } + @if let Some(body) = &data.body_html { + div class="entity-body" { (maud::PreEscaped(body)) } + } + } + } + } +} + +/// Reddit/API import — `POST /ui` with `fetch_entity` returns an SSE stream. +pub fn fetch_entity_panel(item: &ItemId, has_data: bool, fetching: bool) -> Markup { + if !is_fetchable(item) { + return html! {}; + } + let label = if fetching { + "Fetching…" + } else if has_data { + "Fetch more" + } else { + "Fetch from Reddit" + }; + let rpc = template_json_compact(&serde_json::json!({ + "action": "fetch_entity", + "item": item.as_str(), + })) + .expect("fetch_entity rpc template"); + html! { + form method="post" action="/ui" id="fetch-entity-form" class="fetch-entity-form" { + input type="hidden" name=(UI_RPC_FIELD) value=(rpc); + @if fetching { + button type="submit" class="btn-secondary" disabled { (label) } + } @else { + button type="submit" class="btn-secondary" { (label) } + } + } + } +} + +/// Entity card + fetch control (target `#entity-section` for Idiomorph / SSE). +pub fn entity_section(item: &ItemId, node: &NodeState, fetching: bool) -> Markup { + let has_data = node.data.is_some(); + html! { + section id="entity-section" class="demo-panel" { + (entity_panel(node)) + (fetch_entity_panel(item, has_data, fetching)) + } + } +} diff --git a/server/src/fetch/mod.rs b/server/src/fetch/mod.rs new file mode 100644 index 0000000000000000000000000000000000000000..2290f9d3a0f1cbf1806c6339f82a4515c11cc3d3 --- /dev/null +++ b/server/src/fetch/mod.rs @@ -0,0 +1,115 @@ +//! Entity import over `POST /ui` as SSE (Reddit worker in [`crate::reddit`]). + +pub mod html; + +use std::convert::Infallible; +use std::time::Duration; + +use async_stream::stream; +use axum::response::sse::{Event, KeepAlive, Sse}; +use futures_util::Stream; +use serde::Serialize; +use tokio::sync::oneshot; + +use crate::{ + path_types::ItemId, + reddit::FetchJobResult, + reducer::NodeState, + state::AppState, +}; + +pub fn now_ms() -> i64 { + let t = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap_or_default(); + t.as_millis() as i64 +} + +#[derive(Serialize)] +struct SseMorphPayload { + selector: &'static str, + html: String, +} + +fn morph_complete_event(html: maud::Markup) -> Event { + let payload = SseMorphPayload { + selector: "#entity-section", + html: html.into_string(), + }; + let data = serde_json::to_string(&payload).unwrap_or_else(|_| "{}".into()); + Event::default().event("complete").data(data) +} + +/// Stream `fetching` → `complete` / `error` for [`crate::ui_action::HtmlUiAction::FetchEntity`]. +pub fn fetch_entity_stream( + state: AppState, + id: ItemId, +) -> Sse>> { + tracing::debug!(item = %id, "fetch entity stream opened"); + + let stream = stream! { + if id.is_root() { + yield Ok(Event::default().event("error").data("{\"message\":\"nothing to fetch for the root\"}")); + return; + } + + if !crate::reddit::is_fetchable(&id) { + tracing::debug!(item = %id, "fetch stream: not fetchable"); + yield Ok(Event::default().event("error").data("{\"message\":\"this page cannot be fetched from Reddit\"}")); + return; + } + + let fetching_html = { + let tree = state.tree.read().await; + let empty = NodeState::default(); + let node = tree.get(&id).unwrap_or(&empty); + html::entity_section(&id, node, true).into_string() + }; + let fetching_payload = serde_json::json!({ + "selector": "#entity-section", + "html": fetching_html, + }); + yield Ok(Event::default().event("fetching").data(fetching_payload.to_string())); + + let (tx, rx) = oneshot::channel(); + state.reddit.request_fetch(id.clone(), true, Some(tx)); + tracing::debug!(item = %id, "fetch stream: queued reddit job"); + + let result = match rx.await { + Ok(r) => r, + Err(_) => { + tracing::warn!(item = %id, "fetch stream: worker dropped oneshot"); + FetchJobResult::Failed("reddit worker stopped".into()) + } + }; + + tracing::debug!(item = %id, ?result, "fetch stream: job finished"); + + match result { + FetchJobResult::Imported | FetchJobResult::NotFound => { + let tree = state.tree.read().await; + let empty = NodeState::default(); + let node = tree.get(&id).unwrap_or(&empty); + yield Ok(morph_complete_event(html::entity_section(&id, node, false))); + } + FetchJobResult::SkippedCached | FetchJobResult::SkippedDuplicate => { + let tree = state.tree.read().await; + let empty = NodeState::default(); + let node = tree.get(&id).unwrap_or(&empty); + yield Ok(morph_complete_event(html::entity_section(&id, node, false))); + } + FetchJobResult::RateLimited { reset_secs } => { + yield Ok(Event::default().event("error").data( + serde_json::json!({"message": format!("Reddit rate limit — retry in {reset_secs}s")}).to_string(), + )); + } + FetchJobResult::Failed(msg) => { + yield Ok(Event::default().event("error").data( + serde_json::json!({"message": msg}).to_string(), + )); + } + } + }; + + Sse::new(stream).keep_alive(KeepAlive::new().interval(Duration::from_secs(15))) +} diff --git a/server/src/html/mod.rs b/server/src/html/mod.rs index db5b4c7f06b0be64603981166835cde268234f67..9314a7556306ddab969b896dbf4126b542a46722 100644 --- a/server/src/html/mod.rs +++ b/server/src/html/mod.rs @@ -7,10 +7,10 @@ use axum::{ use maud::{html, Markup, DOCTYPE}; use crate::{ + fetch::html::entity_section, form_template::template_json_compact, path_types::ItemId, ranking::{top_bottom, RankedItem}, - reddit::is_fetchable, reducer::{GroupState, NodeState}, state::AppState, ui_action::UI_RPC_FIELD, @@ -149,62 +149,6 @@ pub fn breadcrumb_path(item: &ItemId) -> Markup { } } -fn entity_panel(node: &NodeState) -> Markup { - html! { - @if let Some(data) = &node.data { - div id="entity-panel" class="entity-card" { - h2 { (data.title) } - @if let Some(author) = &data.author { - p class="muted small" { "by " (author) } - } - @if let Some(body) = &data.body_html { - div class="entity-body" { (maud::PreEscaped(body)) } - } - } - } - } -} - -/// Reddit/API import control — only shown on fetchable pages; never auto-fires. -pub fn fetch_entity_panel(item: &ItemId, has_data: bool, fetching: bool) -> Markup { - if !is_fetchable(item) { - return html! {}; - } - let label = if fetching { - "Fetching…" - } else if has_data { - "Fetch more" - } else { - "Fetch from Reddit" - }; - let rpc = template_json_compact(&serde_json::json!({ - "action": "fetch_entity", - "item": item.as_str(), - })) - .expect("fetch_entity rpc template"); - html! { - form method="post" action="/ui" id="fetch-entity-form" class="fetch-entity-form" { - input type="hidden" name=(UI_RPC_FIELD) value=(rpc); - @if fetching { - button type="submit" class="btn-secondary" disabled { (label) } - } @else { - button type="submit" class="btn-secondary" { (label) } - } - } - } -} - -/// Entity card + explicit fetch control (morphed as `#entity-section`). -pub fn entity_section(item: &ItemId, node: &NodeState, fetching: bool) -> Markup { - let has_data = node.data.is_some(); - html! { - section id="entity-section" class="demo-panel" { - (entity_panel(node)) - (fetch_entity_panel(item, has_data, fetching)) - } - } -} - fn rank_list(label: &str, items: &[RankedItem], start_rank: usize) -> Markup { html! { @if !items.is_empty() { diff --git a/server/src/lib.rs b/server/src/lib.rs index 79b173f391a96ae5d0d96fd656e1e8d2dd070d09..0677363e0ed21d244bfa065f7ec728549ddd9e50 100644 --- a/server/src/lib.rs +++ b/server/src/lib.rs @@ -1,6 +1,7 @@ pub mod api; pub mod event_log; pub mod events; +pub mod fetch; pub mod form_template; pub mod html; pub mod parser; diff --git a/server/src/reddit.rs b/server/src/reddit.rs index ff0f01e57b18af878eb5be3efc47204a7673589d..0e6ce32720951a58456852b67fb76a85dd9f4aad 100644 --- a/server/src/reddit.rs +++ b/server/src/reddit.rs @@ -7,12 +7,12 @@ use std::time::{Duration, Instant}; use reqwest::{header, Client, StatusCode}; use serde::Deserialize; use serde_json::Value; -use tokio::sync::{mpsc, RwLock}; +use tokio::sync::{mpsc, oneshot, RwLock}; use crate::{ event_log::EventLog, events::Event, - html::now_ms, + fetch::now_ms, path_types::ItemId, reducer::GlobalTree, }; @@ -22,10 +22,21 @@ pub fn ensure_partial_tree(tree: &mut GlobalTree, id: &ItemId) { tree.ensure_path(id); } +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum FetchJobResult { + Imported, + NotFound, + SkippedDuplicate, + SkippedCached, + RateLimited { reset_secs: u64 }, + Failed(String), +} + pub struct RedditCommand { pub id: ItemId, /// User-initiated fetch bypasses the in-memory "recently fetched" cache. pub force: bool, + pub done: Option>, } #[derive(Clone)] @@ -72,14 +83,29 @@ impl RedditBroker { .build() .expect("reqwest client"); + tracing::debug!( + api_base = %config.api_base, + oauth_base = %config.oauth_base, + oauth = config.creds.is_some(), + "reddit worker started" + ); + tokio::spawn(reddit_worker(rx, tree, event_log, client, config)); Self { tx } } /// Queue a fetch; drops when the channel is full (backpressure). - pub fn request_fetch(&self, id: ItemId, force: bool) { - let _ = self.tx.try_send(RedditCommand { id, force }); + pub fn request_fetch( + &self, + id: ItemId, + force: bool, + done: Option>, + ) { + match self.tx.try_send(RedditCommand { id: id.clone(), force, done }) { + Ok(()) => tracing::debug!(item = %id, force, "reddit fetch queued"), + Err(_) => tracing::warn!(item = %id, "reddit fetch queue full, dropped"), + } } } @@ -95,8 +121,6 @@ impl RedditApiConfig { } impl RedditCredentials { - /// Reddit's OAuth docs call these "client id" and "client secret"; the app - /// registration UI often labels them "app id" / "app secret" — same values. fn from_env() -> Option { let client_id = std::env::var("REDDIT_CLIENT_ID") .or_else(|_| std::env::var("REDDIT_APP_ID")) @@ -128,12 +152,10 @@ pub fn default_user_agent() -> String { }) } -/// True when this node can be loaded from the Reddit JSON API. pub fn is_fetchable(id: &ItemId) -> bool { !map_item_to_reddit_api(id, "https://example.com").is_empty() } -/// Derive UI-facing fields from a stored payload (Reddit-specific when under reddit.com). pub fn entity_view_from_payload(id: &ItemId, payload: &Value) -> Option { if id.as_str().starts_with("reddit.com") { return parse_reddit_view(id, payload); @@ -141,12 +163,17 @@ pub fn entity_view_from_payload(id: &ItemId, payload: &Value) -> Option>, result: FetchJobResult) { + if let Some(tx) = done { + let _ = tx.send(result); + } +} + async fn reddit_worker( mut rx: mpsc::Receiver, tree: Arc>, @@ -168,15 +195,25 @@ async fn reddit_worker( recently_fetched.retain(|_, t| now.duration_since(*t) < cache_ttl); if in_flight.contains(&cmd.id) { + tracing::debug!(item = %cmd.id, "reddit fetch skipped: already in flight"); + notify(cmd.done, FetchJobResult::SkippedDuplicate); continue; } if !cmd.force && recently_fetched.contains_key(&cmd.id) { + tracing::debug!(item = %cmd.id, "reddit fetch skipped: recently fetched cache"); + notify(cmd.done, FetchJobResult::SkippedCached); continue; } in_flight.insert(cmd.id.clone()); let fetch_id = cmd.id.clone(); + let done = cmd.done; + tracing::debug!( + item = %fetch_id, + delay_ms = current_delay.as_millis(), + "reddit fetch starting after delay" + ); tokio::time::sleep(current_delay).await; if let Some(c) = &creds { @@ -184,8 +221,18 @@ async fn reddit_worker( } let token = oauth.as_ref().map(|t| t.access_token.as_str()); + if token.is_some() { + tracing::debug!(item = %fetch_id, "reddit fetch using OAuth bearer"); + } + + let fetch_base = if token.is_some() { + &oauth_base + } else { + &api_base + }; + let outcome = do_fetch(&client, fetch_base, &fetch_id, token).await; - match do_fetch(&client, &api_base, &fetch_id, token).await { + match outcome { Ok(FetchOutcome::Payload(payload)) => { let ts = now_ms(); let event = Event::EntityImported { @@ -193,30 +240,49 @@ async fn reddit_worker( ts, payload: payload.clone(), }; - if let Err(e) = event_log.append(&event).await { - tracing::warn!("event log append failed for {}: {}", fetch_id, e); - } else { - apply_entity_import(&mut *tree.write().await, &fetch_id, payload); - recently_fetched.insert(fetch_id.clone(), Instant::now()); - current_delay = Duration::from_millis(600); + tracing::debug!( + item = %fetch_id, + ts, + payload_keys = ?payload.as_object().map(|o| o.len()), + "reddit fetch got JSON payload, appending event" + ); + match event_log.append(&event).await { + Err(e) => { + tracing::warn!( + item = %fetch_id, + err = %e, + "reddit event log append failed" + ); + notify(done, FetchJobResult::Failed(e.to_string())); + } + Ok(()) => { + apply_entity_import(&mut *tree.write().await, &fetch_id, payload); + recently_fetched.insert(fetch_id.clone(), Instant::now()); + current_delay = Duration::from_millis(600); + tracing::info!(item = %fetch_id, "reddit entity imported"); + notify(done, FetchJobResult::Imported); + } } } Ok(FetchOutcome::NotFound) => { + tracing::debug!(item = %fetch_id, "reddit fetch: not found (no event written)"); recently_fetched.insert(fetch_id.clone(), Instant::now()); + notify(done, FetchJobResult::NotFound); } Ok(FetchOutcome::RateLimited { reset_secs }) => { - let wait = Duration::from_secs(reset_secs.max(1)); tracing::warn!( - "Reddit rate limit for {}; sleeping {}s", - fetch_id, - wait.as_secs() + item = %fetch_id, + reset_secs, + "reddit rate limited" ); - tokio::time::sleep(wait).await; + tokio::time::sleep(Duration::from_secs(reset_secs.max(1))).await; current_delay = (current_delay * 2).min(Duration::from_secs(60)); + notify(done, FetchJobResult::RateLimited { reset_secs }); } Err(e) => { - tracing::warn!("Reddit fetch failed for {}: {}", fetch_id, e); + tracing::warn!(item = %fetch_id, err = %e, "reddit fetch failed"); current_delay = (current_delay * 2).min(Duration::from_secs(60)); + notify(done, FetchJobResult::Failed(e)); } } @@ -238,6 +304,7 @@ async fn ensure_oauth_token( ) -> Option { if let Some(t) = existing { if Instant::now() < t.expires_at - Duration::from_secs(60) { + tracing::debug!("reddit OAuth token still valid"); return Some(t); } } @@ -246,6 +313,7 @@ async fn ensure_oauth_token( "{}/api/v1/access_token", oauth_base.trim_end_matches('/') ); + tracing::debug!(%url, "reddit OAuth token request"); let resp = client .post(&url) @@ -257,13 +325,13 @@ async fn ensure_oauth_token( let resp = match resp { Ok(r) => r, Err(e) => { - tracing::warn!("Reddit OAuth token request failed: {e}"); + tracing::warn!("reddit OAuth token request failed: {e}"); return None; } }; if !resp.status().is_success() { - tracing::warn!("Reddit OAuth token HTTP {}", resp.status()); + tracing::warn!("reddit OAuth token HTTP {}", resp.status()); return None; } @@ -276,11 +344,12 @@ async fn ensure_oauth_token( let body: TokenResponse = match resp.json().await { Ok(b) => b, Err(e) => { - tracing::warn!("Reddit OAuth token parse failed: {e}"); + tracing::warn!("reddit OAuth token parse failed: {e}"); return None; } }; + tracing::debug!(expires_in = body.expires_in, "reddit OAuth token acquired"); Some(OAuthToken { access_token: body.access_token, expires_at: Instant::now() + Duration::from_secs(body.expires_in), @@ -295,35 +364,72 @@ async fn do_fetch( ) -> Result { let url = map_item_to_reddit_api(id, api_base); if url.is_empty() { + tracing::debug!(item = %id, "reddit do_fetch: no API URL for item"); return Ok(FetchOutcome::NotFound); } + tracing::debug!(item = %id, %url, bearer = bearer.is_some(), "reddit HTTP GET"); + let mut req = client.get(&url); if let Some(token) = bearer { req = req.bearer_auth(token); } - let resp = req.send().await.map_err(|e| e.to_string())?; + let resp = req.send().await.map_err(|e| { + tracing::debug!(item = %id, %url, err = %e, "reddit HTTP transport error"); + e.to_string() + })?; + + let status = resp.status(); + tracing::debug!( + item = %id, + %url, + %status, + remaining = ?rate_limit_remaining(&resp), + reset = ?rate_limit_reset_secs(&resp), + "reddit HTTP response" + ); - if resp.status() == StatusCode::TOO_MANY_REQUESTS { + if status == StatusCode::TOO_MANY_REQUESTS { let reset = rate_limit_reset_secs(&resp); return Ok(FetchOutcome::RateLimited { reset_secs: reset }); } - if resp.status() == StatusCode::SERVICE_UNAVAILABLE { - return Err("Reddit unavailable (503)".to_string()); + if status == StatusCode::SERVICE_UNAVAILABLE { + return Err("Reddit unavailable (503)".into()); } - if !resp.status().is_success() { + if !status.is_success() { + let body = resp.text().await.unwrap_or_default(); + tracing::debug!( + item = %id, + %status, + body_len = body.len(), + body_prefix = %body.chars().take(240).collect::(), + "reddit non-success body" + ); return Ok(FetchOutcome::NotFound); } if rate_limit_remaining(&resp) == Some(0) { let reset = rate_limit_reset_secs(&resp); + tracing::debug!(item = %id, reset_secs = reset, "reddit headers: rate limit exhausted"); return Ok(FetchOutcome::RateLimited { reset_secs: reset }); } - let payload: Value = resp.json().await.map_err(|e| e.to_string())?; + let text = resp.text().await.map_err(|e| e.to_string())?; + tracing::debug!(item = %id, bytes = text.len(), "reddit response body received"); + + let payload: Value = serde_json::from_str(&text).map_err(|e| { + tracing::debug!( + item = %id, + err = %e, + body_prefix = %text.chars().take(240).collect::(), + "reddit JSON parse failed" + ); + format!("invalid JSON: {e}") + })?; + Ok(FetchOutcome::Payload(payload)) } @@ -344,7 +450,6 @@ fn rate_limit_reset_secs(resp: &reqwest::Response) -> u64 { .unwrap_or(5) } -/// Map canonical item id to a Reddit JSON API URL under `api_base`. pub fn map_item_to_reddit_api(id: &ItemId, api_base: &str) -> String { let path = id.as_str(); if !path.starts_with("reddit.com/") && path != "reddit.com" { @@ -445,26 +550,6 @@ mod tests { map_item_to_reddit_api(&id, "https://www.reddit.com"), "https://www.reddit.com/r/rust/about.json?raw_json=1" ); - assert_eq!( - map_item_to_reddit_api(&id, "http://127.0.0.1:9999"), - "http://127.0.0.1:9999/r/rust/about.json?raw_json=1" - ); - } - - #[test] - fn map_post_url() { - let id = ItemId::parse("reddit.com/r/amitheasshole/comments/1trnvdl").unwrap(); - assert_eq!( - map_item_to_reddit_api(&id, "https://www.reddit.com"), - "https://www.reddit.com/r/amitheasshole/comments/1trnvdl.json?raw_json=1" - ); - } - - #[test] - fn is_fetchable_reddit_sub() { - let id = ItemId::parse("reddit.com/r/rust").unwrap(); - assert!(is_fetchable(&id)); - assert!(!is_fetchable(&ItemId::opaque("example.com/x"))); } #[test] @@ -477,19 +562,5 @@ mod tests { ) .unwrap(); assert_eq!(entity.title, "The Rust Programming Language"); - assert!(entity.body_html.as_ref().is_some_and(|b| b.contains("Rust"))); - } - - #[test] - fn parse_post_fixture() { - let json = r#"[{"kind":"Listing","data":{"children":[{"kind":"t3","data":{"title":"AITA","author":"op","selftext_html":"<p>hi</p>","thumbnail":"https://b.thumbs.redditmedia.com/x.jpg"}}]}}]"#; - let v: Value = serde_json::from_str(json).unwrap(); - let entity = entity_view_from_payload( - &ItemId::parse("reddit.com/r/x/comments/abc").unwrap(), - &v, - ) - .unwrap(); - assert_eq!(entity.title, "AITA"); - assert_eq!(entity.author.as_deref(), Some("op")); } } diff --git a/server/src/state.rs b/server/src/state.rs index d71d1079a8f486cdab15795384aef0b81b32544d..e7ff9f663e45b5e168d6a3869948e3bd890966d4 100644 --- a/server/src/state.rs +++ b/server/src/state.rs @@ -143,9 +143,9 @@ impl AppState { Ok(()) } - /// User-initiated Reddit/API import (via "Fetch more" — never on paste or navigate). - pub fn queue_entity_fetch(&self, id: ItemId) { - self.reddit.request_fetch(id, true); + /// User-initiated Reddit/API import (SSE / fetch module only). + pub fn queue_entity_fetch(&self, id: ItemId, done: Option>) { + self.reddit.request_fetch(id, true, done); } pub async fn record_vote( diff --git a/server/src/ui_action.rs b/server/src/ui_action.rs index 3d6a49a2a3fb950752827efe1f5a049308f09baf..ef9ac873fc0442e0b960f144309c8e91124d7884 100644 --- a/server/src/ui_action.rs +++ b/server/src/ui_action.rs @@ -26,7 +26,7 @@ pub enum HtmlUiAction { ParseQuery { query: String, }, - /// Fetch upstream entity data for the current page (explicit user action only). + /// Import entity data; `POST /ui` responds with `text/event-stream` (not JS). FetchEntity { item: String, }, diff --git a/server/static/sorter_ui.js b/server/static/sorter_ui.js index d9b016f197547f37ca4b2bcdd7ee6b673d0fb3f0..fe7bfc657c0fccae5c596d005b0e807abefec677 100644 --- a/server/static/sorter_ui.js +++ b/server/static/sorter_ui.js @@ -1,5 +1,5 @@ /** - * sorter2 web UI: fetch/eval for POST /ui. No product logic here. + * sorter2 web UI: POST /ui returns JS (morph) or SSE (entity fetch). */ (function () { function evalJs(js) { @@ -8,15 +8,98 @@ } } + function morphSelector(selector, html) { + var el = document.querySelector(selector); + if (el && typeof Idiomorph !== 'undefined') { + Idiomorph.morph(el, html); + } + } + + function handleSseEvent(eventType, data, form) { + if (eventType === 'fetching' || eventType === 'complete') { + try { + var msg = JSON.parse(data); + morphSelector(msg.selector || '#entity-section', msg.html); + } catch (err) { + console.warn('fetch morph parse', err); + } + } + if (eventType === 'complete' || eventType === 'error') { + var btn = form && form.querySelector('button[type="submit"]'); + if (btn) btn.disabled = false; + } + if (eventType === 'error') { + try { + var err = JSON.parse(data); + console.warn('fetch error:', err.message || data); + } catch (_e) { + console.warn('fetch error:', data); + } + } + } + + function consumeSseStream(response, form) { + var reader = response.body.getReader(); + var decoder = new TextDecoder(); + var buffer = ''; + var eventType = ''; + var dataLines = []; + + function dispatch() { + if (!eventType && dataLines.length === 0) return; + handleSseEvent(eventType || 'message', dataLines.join('\n'), form); + eventType = ''; + dataLines = []; + } + + function pump() { + return reader.read().then(function (chunk) { + if (chunk.done) { + dispatch(); + return; + } + buffer += decoder.decode(chunk.value, { stream: true }); + var parts = buffer.split('\n'); + buffer = parts.pop() || ''; + for (var i = 0; i < parts.length; i++) { + var line = parts[i].replace(/\r$/, ''); + if (line === '') { + dispatch(); + } else if (line.indexOf('event:') === 0) { + eventType = line.slice(6).trim(); + } else if (line.indexOf('data:') === 0) { + dataLines.push(line.slice(5).trim()); + } + } + return pump(); + }); + } + + return pump(); + } + function postUiForm(form) { + var btn = form.querySelector('button[type="submit"]'); + if (form.id === 'fetch-entity-form' && btn) { + btn.disabled = true; + } return fetch(form.action, { method: 'POST', body: new URLSearchParams(new FormData(form)), headers: { 'Content-Type': 'application/x-www-form-urlencoded' }, credentials: 'same-origin', }).then(function (resp) { - return resp.text(); - }).then(evalJs); + var ct = resp.headers.get('content-type') || ''; + if (ct.indexOf('text/event-stream') !== -1) { + return consumeSseStream(resp, form); + } + return resp.text().then(evalJs); + }).catch(function (err) { + if (form.id === 'fetch-entity-form' && btn) { + btn.disabled = false; + } + console.warn('POST /ui failed', err); + }); } function initSorterUi() { diff --git a/test/reddit_import.clj b/test/reddit_import.clj index 54edaedeef08718c8f184d6aa468d8e6415068c7..17f2ee39e1498b0b6bb6a9d8a7e35608b0bd1b32 100644 --- a/test/reddit_import.clj +++ b/test/reddit_import.clj @@ -44,10 +44,12 @@ (do (Thread/sleep 200) (recur)) false)))))) -(defn- curl-post-ui [base rpc-json] +(defn- curl-fetch-ui-sse [base item] (process/shell {:out :string :err :string} - "curl" "-sf" "-X" "POST" (str base "/ui") - "--data-urlencode" (str "__rpc__=" rpc-json))) + "curl" "-sfN" "--max-time" "20" + "-X" "POST" (str base "/ui") + "--data-urlencode" + (str "__rpc__={\"action\":\"fetch_entity\",\"item\":\"" item "\"}"))) (defn- wait-event-log [path ms] (let [deadline (+ (System/currentTimeMillis) ms)] @@ -98,11 +100,12 @@ "curl" "-sf" browse-url))] (is (str/includes? before "Fetch from Reddit")) (is (not (str/includes? before "The Rust Programming Language"))) - (let [rpc "{\"action\":\"fetch_entity\",\"item\":\"reddit.com/r/rust\"}" - post (curl-post-ui app-base rpc) - log-path (str data-dir "/events.jsonl")] - (is (zero? (:exit post)) "fetch_entity POST succeeds") - (is (wait-event-log log-path 10000) "event log written") + (let [log-path (str data-dir "/events.jsonl") + sse (curl-fetch-ui-sse app-base "reddit.com/r/rust")] + (is (zero? (:exit sse)) "POST /ui fetch_entity SSE succeeds") + (is (str/includes? (:out sse) "event: complete")) + (is (str/includes? (:out sse) "The Rust Programming Language")) + (is (wait-event-log log-path 2000) "event log written") (let [after (:out (process/shell {:out :string :err :string} "curl" "-sf" browse-url)) log (slurp (io/file log-path))]