You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [af73743d] Replace GroupState with ScopeVotes and derive edges at ranking time. Store only uuid_votes and recent_votes per scope; rank centrality and pair logic rebuild edge weights on demand instead of maintaining cached state. Co-authored-by: Cursor Side A — unified diff (full patch): diff --git a/server/src/events.rs b/server/src/events.rs index 8a166d49b4f26835fbc2b58cb1f4bdbf002763b8..015208311f6c5c23a0e8aab068d002a69c89c4e1 100644 --- a/server/src/events.rs +++ b/server/src/events.rs @@ -43,7 +43,7 @@ pub enum ViewEvent { #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(tag = "type", rename_all = "snake_case")] pub enum Event { - /// Pairwise comparison vote (replayed into the parent node's [`crate::reducer::GroupState`] on boot). + /// Pairwise comparison vote (replayed into the parent node's [`crate::reducer::ScopeVotes`] on boot). /// `scope` is the parent [`crate::path_types::ItemId`] string; empty string is the tree root. VoteRecorded { ts: i64, diff --git a/server/src/html/mod.rs b/server/src/html/mod.rs index 4eff2e19ed4d303ff8e80c1eabd8a15b4990e643..1e2e7a06856d8a62378741aaf5ed94a4ffed337e 100644 --- a/server/src/html/mod.rs +++ b/server/src/html/mod.rs @@ -13,7 +13,7 @@ use crate::{ form_template::template_json_compact, path_types::ItemId, ranking::{ - connected_components_from_voted_pairs, ranked_items_subset, RankedItem, MAX_ITERS, TOL, + ranked_items_subset, scope_components, RankedItem, MAX_ITERS, TOL, }, reducer::{GlobalTree, NodeState}, state::AppState, @@ -397,10 +397,9 @@ pub fn ranking_panel_with_highlights( tree: &GlobalTree, highlighted: &HashSet, ) -> Markup { - let group = &node.local_ranking; - let n = group.idx_to_item.len(); - let (comps, _isolates) = - connected_components_from_voted_pairs(n, group.voted_pairs.iter().copied()); + let scope = &node.votes; + let (comps, _isolates, _) = + scope_components(scope); // Each connected component of voted items is its own ranking; isolated and // never-voted children fall into the "unranked" bucket below. @@ -410,7 +409,7 @@ pub fn ranking_panel_with_highlights( if comp.len() < 2 { continue; } - let ranked = ranked_items_subset(group, comp, MAX_ITERS, TOL); + let ranked = ranked_items_subset(scope, comp, MAX_ITERS, TOL); for r in &ranked { ranked_ids.insert(r.item.clone()); } diff --git a/server/src/html/vote.rs b/server/src/html/vote.rs index bf82aef3ad9c5f4e9e877c47dab27beb29a80b8f..3aa00c417c89a9cab3417c650b50ed7c73f08e20 100644 --- a/server/src/html/vote.rs +++ b/server/src/html/vote.rs @@ -14,7 +14,7 @@ use crate::{ html::{ranking_panel_with_highlights, scope_theme_style, JsBuilder}, pair::{children_of, resolve_pair, suggest_next_pair_in_pool}, path_types::ItemId, - reducer::{GlobalTree, GroupState, NodeState, VoteData}, + reducer::{GlobalTree, NodeState, ScopeVotes, VoteData}, state::{parse_item_param, AppState}, ui_action::UI_RPC_FIELD, }; @@ -68,8 +68,8 @@ fn ratios_for_page(v: &VoteData, page_left: &ItemId, page_right: &ItemId) -> (i3 } } -fn edge_votes(group: &GroupState, left: &ItemId, right: &ItemId) -> Vec { - group +fn edge_votes(scope: &ScopeVotes, left: &ItemId, right: &ItemId) -> Vec { + scope .recent_votes .iter() .filter(|v| { @@ -113,11 +113,11 @@ fn slider_value_from_ratios(r_left: i32, r_right: i32) -> i32 { fn vote_edge_history( tree: &GlobalTree, - group: &GroupState, + scope: &ScopeVotes, left: &ItemId, right: &ItemId, ) -> Markup { - let mut votes = edge_votes(group, left, right); + let mut votes = edge_votes(scope, left, right); votes.sort_by(|a, b| b.ts.cmp(&a.ts)); let legend_left = child_title(tree, left); let legend_right = child_title(tree, right); @@ -228,9 +228,9 @@ pub(crate) fn vote_recorded_morph( ) -> JsBuilder { let pool = children_of(tree, parent); let empty = NodeState::default(); - let group = tree.get(parent).unwrap_or(&empty).local_ranking.clone(); - let edge_history = vote_edge_history(tree, &group, left, right); - let next_pair = suggest_next(&group, left, right, &pool); + let scope = tree.get(parent).unwrap_or(&empty).votes.clone(); + let edge_history = vote_edge_history(tree, &scope, left, right); + let next_pair = suggest_next(&scope, left, right, &pool); let actions = vote_compare_actions(parent, next_pair.as_ref()); let sidebar = vote_ranking_sidebar(tree, parent, left, right); JsBuilder::new() @@ -252,12 +252,12 @@ fn vote_compare_item_card(tree: &GlobalTree, item: &ItemId, side_class: &str) -> } fn suggest_next( - group: &GroupState, + scope: &ScopeVotes, left: &ItemId, right: &ItemId, pool: &[ItemId], ) -> Option<(ItemId, ItemId)> { - suggest_next_pair_in_pool(group, pool, Some((left, right))) + suggest_next_pair_in_pool(scope, pool, Some((left, right))) } pub async fn vote_page( @@ -284,9 +284,9 @@ pub async fn vote_page( }; let pool = children_of(&tree, &parent); - let group = &parent_node.local_ranking; - let next_pair = suggest_next(group, &left, &right, &pool); - let edge_history = vote_edge_history(&tree, group, &left, &right); + let scope = &parent_node.votes; + let next_pair = suggest_next(&scope, &left, &right, &pool); + let edge_history = vote_edge_history(&tree, &scope, &left, &right); let rpc_json = template_json_compact(&serde_json::json!({ "action": "record_vote", @@ -388,8 +388,8 @@ mod polarity_tests { let mut tree = GlobalTree::new(); tree.apply_vote(&parent, vote, TEST_ACTOR_UUID); - let group = &tree.get(&parent).unwrap().local_ranking; - let ranked = ranked_items(group); + let scope = &tree.get(&parent).unwrap().votes; + let ranked = ranked_items(scope); assert_eq!( ranked[0].item, left, "left item should rank first when ratio favours the left" diff --git a/server/src/pair.rs b/server/src/pair.rs index 42a1b1eb2adf16730d34d0fe23c13d5a75d7ba27..9873295c51526726089875bbfd2f97d7faa91872 100644 --- a/server/src/pair.rs +++ b/server/src/pair.rs @@ -11,8 +11,8 @@ use std::collections::{HashMap, HashSet}; use crate::{ path_types::ItemId, - ranking::{connected_components_from_voted_pairs, ranked_items}, - reducer::{GlobalTree, GroupState}, + ranking::{pair_is_voted, ranked_items, scope_components}, + reducer::{GlobalTree, ScopeVotes}, }; fn pairs_match(a: &ItemId, b: &ItemId, x: &ItemId, y: &ItemId) -> bool { @@ -23,26 +23,15 @@ fn pair_excluded(a: &ItemId, b: &ItemId, exclude: Option<(&ItemId, &ItemId)>) -> exclude.is_some_and(|(x, y)| pairs_match(a, b, x, y)) } -fn pair_is_voted(group: &GroupState, a: &ItemId, b: &ItemId) -> bool { - let Some(&ai) = group.item_to_idx.get(a) else { - return false; - }; - let Some(&bi) = group.item_to_idx.get(b) else { - return false; - }; - let (i, j) = if ai < bi { (ai, bi) } else { (bi, ai) }; - group.voted_pairs.contains(&(i, j)) -} struct ComponentLayout { ids: HashMap, established: HashSet, } -fn component_layout(group: &GroupState, pool: &[ItemId]) -> ComponentLayout { - let n = group.idx_to_item.len(); - let (comps, isolates) = - connected_components_from_voted_pairs(n, group.voted_pairs.iter().copied()); +fn component_layout(scope: &ScopeVotes, pool: &[ItemId]) -> ComponentLayout { + let (comps, isolates, idx_to_item) = scope_components(scope); + let n = idx_to_item.len(); let mut established = HashSet::new(); let mut ids: HashMap = HashMap::new(); @@ -52,14 +41,14 @@ fn component_layout(group: &GroupState, pool: &[ItemId]) -> ComponentLayout { } for &idx in comp { if idx < n { - ids.insert(group.idx_to_item[idx].clone(), comp_idx); + ids.insert(idx_to_item[idx].clone(), comp_idx); } } } let mut next = comps.len(); for &idx in &isolates { if idx < n { - ids.insert(group.idx_to_item[idx].clone(), next); + ids.insert(idx_to_item[idx].clone(), next); next += 1; } } @@ -121,7 +110,7 @@ fn established_groups_in_pool<'a>( groups } -fn ranked_pool_order(group: &GroupState, pool: &[ItemId]) -> Vec { +fn ranked_pool_order(group: &ScopeVotes, pool: &[ItemId]) -> Vec { let pool_set: HashSet<_> = pool.iter().collect(); ranked_items(group) .into_iter() @@ -132,7 +121,7 @@ fn ranked_pool_order(group: &GroupState, pool: &[ItemId]) -> Vec { /// Walk 1↔2, 2↔3, …; optional `require_unvoted` skips voted edges. fn zip_adjacent_pair( - group: &GroupState, + group: &ScopeVotes, order: &[ItemId], exclude: Option<(&ItemId, &ItemId)>, require_unvoted: bool, @@ -153,7 +142,7 @@ fn zip_adjacent_pair( /// Grow the voted graph toward one component (no rank centrality). fn suggest_grow_pair( - group: &GroupState, + group: &ScopeVotes, pool: &[ItemId], layout: &ComponentLayout, exclude: Option<(&ItemId, &ItemId)>, @@ -216,7 +205,7 @@ fn suggest_grow_pair( /// Pick the next pair to vote on within `pool`. pub fn suggest_next_pair_in_pool( - group: &GroupState, + group: &ScopeVotes, pool: &[ItemId], exclude: Option<(&ItemId, &ItemId)>, ) -> Option<(ItemId, ItemId)> { @@ -315,7 +304,7 @@ pub fn resolve_pair( (None, None) => { let group = tree .get(parent) - .map(|n| &n.local_ranking) + .map(|n| &n.votes) .cloned() .unwrap_or_default(); suggest_next_pair_in_pool(&group, &children, None).ok_or(PairError::NoPair) @@ -398,7 +387,7 @@ mod tests { "https://reddit.com/r/rust/b", ], ); - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); assert!(!pair_is_voted(&group, &pool[0], &pool[1])); assert!(suggest_next_pair_in_pool(&group, &pool, None).is_some()); @@ -417,7 +406,7 @@ mod tests { ); let vote = test_vote(1, "https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b", 2, 1); apply(&mut tree, &parent, vote); - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let (l, r) = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let voted_ab = (l.as_str() == "https://reddit.com/r/rust/a" && r.as_str() == "https://reddit.com/r/rust/b") @@ -441,7 +430,7 @@ mod tests { let cd = test_vote(2, "https://reddit.com/r/rust/c", "https://reddit.com/r/rust/d", 2, 1); apply(&mut tree, &parent, ab); apply(&mut tree, &parent, cd); - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); @@ -467,7 +456,7 @@ mod tests { ); let ab = test_vote(1, "https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b", 2, 1); apply(&mut tree, &parent, ab); - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); @@ -496,7 +485,7 @@ mod tests { ); let ab = test_vote(1, "https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b", 2, 1); apply(&mut tree, &parent, ab); - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); @@ -522,7 +511,7 @@ mod tests { let v = test_vote(1, a, b, l, r); apply(&mut tree, &parent, v); } - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); @@ -550,7 +539,7 @@ mod tests { let v = test_vote(1, a, b, l, r); apply(&mut tree, &parent, v); } - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); diff --git a/server/src/projection_store.rs b/server/src/projection_store.rs index 27fc0bb0519d9f035dceec1fc8b0fa74b00b6b40..8c1a466183a96173fe52b144fb67c2454b715fbd 100644 --- a/server/src/projection_store.rs +++ b/server/src/projection_store.rs @@ -214,7 +214,7 @@ mod tests { let loaded = store.load_tree().unwrap(); let root = loaded.get(&ItemId::root()).unwrap(); - assert_eq!(root.local_ranking.idx_to_item.len(), 2); + assert_eq!(crate::ranking::ranked_items(&root.votes).len(), 2); assert!(root.children.contains(&ItemId::opaque("alpha"))); } diff --git a/server/src/ranking.rs b/server/src/ranking.rs index 1fc3298d2b2864e9711cd262a034677e45c7090c..cc4de6da11b02135e5e6b1d2a68646cb77870b9d 100644 --- a/server/src/ranking.rs +++ b/server/src/ranking.rs @@ -1,7 +1,7 @@ -use std::collections::{HashMap, HashSet}; +use std::collections::{BTreeSet, HashMap, HashSet}; use crate::path_types::ItemId; -use crate::reducer::GroupState; +use crate::reducer::{canonical_pair_ids, ScopeVotes}; #[derive(Debug, Clone)] pub struct RankedItem { @@ -9,15 +9,76 @@ pub struct RankedItem { pub score: f64, } -/// Power-iteration cap and convergence tolerance for rank centrality. pub const MAX_ITERS: usize = 10_000; pub const TOL: f64 = 1e-8; -/// Compute connected components over the voted-pairs graph (treated as undirected). -/// -/// Returns: -/// - `components`: each component is a sorted list of node indices, excluding isolates. -/// - `isolates`: sorted list of node indices with degree 0 (no voted pairs). +pub fn item_index(scope: &ScopeVotes) -> (HashMap, Vec) { + let mut item_strs: BTreeSet = BTreeSet::new(); + for vote in scope.uuid_votes.values() { + item_strs.insert(vote.a.as_str().to_string()); + item_strs.insert(vote.b.as_str().to_string()); + } + let mut idx_to_item: Vec = Vec::with_capacity(item_strs.len()); + let mut item_to_idx: HashMap = HashMap::with_capacity(item_strs.len()); + for s in item_strs { + let id = ItemId::from_storage(&s).unwrap_or_else(|| ItemId::opaque(&s)); + let idx = idx_to_item.len(); + item_to_idx.insert(id.clone(), idx); + idx_to_item.push(id); + } + (item_to_idx, idx_to_item) +} + +pub fn edges_from_scope(scope: &ScopeVotes) -> HashMap<(usize, usize), f64> { + let (item_to_idx, _) = item_index(scope); + let mut edges: HashMap<(usize, usize), f64> = HashMap::new(); + for vote in scope.uuid_votes.values() { + let Some(&ai) = item_to_idx.get(&vote.a) else { + continue; + }; + let Some(&bi) = item_to_idx.get(&vote.b) else { + continue; + }; + let w_a = vote.ratio_left as f64 * vote.trust_weight; + let w_b = vote.ratio_right as f64 * vote.trust_weight; + if w_a > 0.0 { + *edges.entry((bi, ai)).or_insert(0.0) += w_a; + } + if w_b > 0.0 { + *edges.entry((ai, bi)).or_insert(0.0) += w_b; + } + } + edges +} + +pub fn edge_weight_sum(scope: &ScopeVotes) -> f64 { + edges_from_scope(scope).values().sum() +} + +pub fn voted_pair_indices(scope: &ScopeVotes) -> HashSet<(usize, usize)> { + let (item_to_idx, _) = item_index(scope); + let mut pairs = HashSet::new(); + for vote in scope.uuid_votes.values() { + let Some(&ai) = item_to_idx.get(&vote.a) else { + continue; + }; + let Some(&bi) = item_to_idx.get(&vote.b) else { + continue; + }; + let (i, j) = if ai < bi { (ai, bi) } else { (bi, ai) }; + pairs.insert((i, j)); + } + pairs +} + +pub fn pair_is_voted(scope: &ScopeVotes, a: &ItemId, b: &ItemId) -> bool { + let (lo, hi) = canonical_pair_ids(a, b); + scope + .uuid_votes + .keys() + .any(|(_, l, h)| l == &lo && h == &hi) +} + pub fn connected_components_from_voted_pairs( n: usize, voted_pairs: impl Iterator, @@ -63,20 +124,25 @@ pub fn connected_components_from_voted_pairs( (comps, isolates) } -/// Compute rank-centrality scores for the whole group and return items sorted -/// by score (descending). Recomputed fresh from the edge set on every call — -/// there is no score cache. -pub fn ranked_items(group: &GroupState) -> Vec { - let n = group.idx_to_item.len(); - let scores = - compute_scores_from_edges(n, group.edges.iter().map(|(&k, &w)| (k, w)), MAX_ITERS, TOL); +pub fn scope_components(scope: &ScopeVotes) -> (Vec>, Vec, Vec) { + let (_, idx_to_item) = item_index(scope); + let n = idx_to_item.len(); + let pairs = voted_pair_indices(scope); + let (comps, isolates) = connected_components_from_voted_pairs(n, pairs.into_iter()); + (comps, isolates, idx_to_item) +} - let mut items: Vec = group - .idx_to_item - .iter() +pub fn ranked_items(scope: &ScopeVotes) -> Vec { + let (_, idx_to_item) = item_index(scope); + let n = idx_to_item.len(); + let edges = edges_from_scope(scope); + let scores = compute_scores_from_edges(n, edges.into_iter(), MAX_ITERS, TOL); + + let mut items: Vec = idx_to_item + .into_iter() .enumerate() .map(|(i, item)| RankedItem { - item: item.clone(), + item, score: *scores.get(i).unwrap_or(&0.0), }) .collect(); @@ -89,11 +155,8 @@ pub fn ranked_items(group: &GroupState) -> Vec { items } -/// Highest- and lowest-ranked items for a group. Returns up to `k` items from -/// each end with no overlap. If the group has `2*k` items or fewer, `top` holds -/// the full ranking and `bottom` is empty (so nothing is shown twice). -pub fn top_bottom(group: &GroupState, k: usize) -> (Vec, Vec) { - let items = ranked_items(group); +pub fn top_bottom(scope: &ScopeVotes, k: usize) -> (Vec, Vec) { + let items = ranked_items(scope); if k == 0 || items.len() <= 2 * k { return (items, Vec::new()); } @@ -115,7 +178,6 @@ pub fn compute_scores_from_edges( return vec![1.0]; } - // Collect raw edges into a map for pairwise normalization. let mut raw: HashMap<(usize, usize), f64> = HashMap::new(); for ((src, dst), w) in edges { if src >= n || dst >= n || w <= 0.0 { @@ -124,9 +186,6 @@ pub fn compute_scores_from_edges( *raw.entry((src, dst)).or_insert(0.0) += w; } - // Pairwise normalization: a_ij = A_ij / (A_ij + A_ji). - // This ensures repeated votes on the same pair don't inflate influence - // beyond what the ratio implies. let keys: Vec<(usize, usize)> = raw.keys().copied().collect(); let mut normalized: HashMap<(usize, usize), f64> = HashMap::new(); for (i, j) in keys { @@ -145,17 +204,6 @@ pub fn compute_scores_from_edges( } } - // Rank Centrality (Negahban, Oh, Shah 2012, §3.1): - // P_ij = (1/d_max) * A_ij for i ≠ j compared - // P_ii = 1 - (1/d_max) * Σ_k A_ik - // where d_i is the *degree* (number of distinct neighbors compared) and - // d_max = max_i d_i. Using the unweighted degree — not the sum of - // pairwise-normalized weights — is what guarantees aperiodicity: it - // forces P_ii > 0 for every non-maximum-degree node, and for max-degree - // nodes whenever any neighbor weight is below 1 (i.e. not a unanimous - // loss). Without this, regular comparison graphs (e.g. a pure star at - // ratio 2:1) produce a bipartite chain that oscillates instead of - // converging — see issue #146. let mut out_edges: Vec> = vec![Vec::new(); n]; let mut neighbors: Vec> = vec![HashSet::new(); n]; @@ -213,11 +261,8 @@ pub fn compute_scores_from_edges( scores } -/// Rank-centrality within a subset of items (an induced subgraph), using the group's aggregated edges. -/// -/// `idxs` are indices into `group.idx_to_item`. The returned items use the original item names. pub fn ranked_items_subset( - group: &GroupState, + scope: &ScopeVotes, idxs: &[usize], max_iters: usize, tol: f64, @@ -226,13 +271,15 @@ pub fn ranked_items_subset( return vec![]; } - // Map original idx -> compact idx [0..m) + let (_, idx_to_item) = item_index(scope); + let edges = edges_from_scope(scope); + let mut map: HashMap = HashMap::with_capacity(idxs.len()); for (j, &i) in idxs.iter().enumerate() { map.insert(i, j); } - let edges_iter = group.edges.iter().filter_map(|(&(src, dst), &w)| { + let edges_iter = edges.into_iter().filter_map(|((src, dst), w)| { let s = *map.get(&src)?; let d = *map.get(&dst)?; Some(((s, d), w)) @@ -240,12 +287,11 @@ pub fn ranked_items_subset( let scores = compute_scores_from_edges(idxs.len(), edges_iter, max_iters, tol); - // Filter out entries where idx_to_item doesn't have the slot (shouldn't happen, but be safe). let mut items: Vec = idxs .iter() .enumerate() .filter_map(|(j, &orig)| { - let item = group.idx_to_item.get(orig)?.clone(); + let item = idx_to_item.get(orig)?.clone(); Some(RankedItem { item, score: *scores.get(j).unwrap_or(&0.0), @@ -261,8 +307,8 @@ pub fn ranked_items_subset( items } -pub fn group_summary_scores(group: &GroupState) -> HashMap { - ranked_items(group) +pub fn group_summary_scores(scope: &ScopeVotes) -> HashMap { + ranked_items(scope) .into_iter() .map(|r| (r.item, r.score)) .collect() @@ -274,32 +320,26 @@ mod tests { use crate::identity::{DEFAULT_PSEUDONYM, TEST_ACTOR_UUID}; use crate::reducer::VoteData; - fn mk_group() -> GroupState { - GroupState::new() + fn mk_scope() -> ScopeVotes { + ScopeVotes::default() } fn vote(ts: i64, a: &str, b: &str, l: i32, r: i32) -> VoteData { VoteData::from_event(ts, a, b, l, r, DEFAULT_PSEUDONYM.to_string(), 1.0).unwrap() } - fn apply(g: &mut GroupState, v: VoteData) { - g.apply_vote(v, TEST_ACTOR_UUID); + fn apply(scope: &mut ScopeVotes, v: VoteData) { + scope.apply_vote(v, TEST_ACTOR_UUID); } - /// Regression for issue #146: pure forward star at default `>` ratio (2:1). - /// Under the old (sum-of-weights) divisor every node had P_ii = 0 and the - /// chain was bipartite; power iteration oscillated and returned the - /// uniform initial distribution after an even number of steps. Using the - /// paper's degree-based d_max gives every node a positive self-loop and - /// the chain converges to the correct stationary distribution. #[test] fn star_topology_winner_at_top_via_subset() { - let mut g = mk_group(); - g.apply_vote(vote(1, "zebra", "alpha", 2, 1), TEST_ACTOR_UUID); - g.apply_vote(vote(2, "zebra", "beta", 2, 1), TEST_ACTOR_UUID); + let mut scope = mk_scope(); + apply(&mut scope, vote(1, "zebra", "alpha", 2, 1)); + apply(&mut scope, vote(2, "zebra", "beta", 2, 1)); - let mut items: Vec<(usize, String)> = g - .idx_to_item + let (_, idx_to_item) = item_index(&scope); + let mut items: Vec<(usize, String)> = idx_to_item .iter() .enumerate() .map(|(i, it)| (i, it.as_str().to_string())) @@ -307,78 +347,51 @@ mod tests { items.sort_by(|a, b| a.1.cmp(&b.1)); let idxs: Vec = items.iter().map(|(i, _)| *i).collect(); - let ranked = ranked_items_subset(&g, &idxs, 10000, 1e-8); - for r in &ranked { - eprintln!("{}: {}", r.item.as_str(), r.score); - } - assert_eq!( - ranked[0].item.as_str(), - "zebra", - "zebra won both votes and should rank #1" - ); + let ranked = ranked_items_subset(&scope, &idxs, 10000, 1e-8); + assert_eq!(ranked[0].item.as_str(), "zebra"); } #[test] fn top_bottom_splits_ends_without_overlap() { - let mut g = mk_group(); - // Chain a > b > c > d > e > f so ranks are well separated. + let mut scope = mk_scope(); for (hi, lo) in [("a", "b"), ("b", "c"), ("c", "d"), ("d", "e"), ("e", "f")] { - apply(&mut g, vote(1, hi, lo, 2, 1)); + apply(&mut scope, vote(1, hi, lo, 2, 1)); } - let (top, bottom) = top_bottom(&g, 2); + let (top, bottom) = top_bottom(&scope, 2); assert_eq!(top.len(), 2); assert_eq!(bottom.len(), 2); - // No overlap between the two ends. for t in &top { assert!(bottom.iter().all(|b| b.item != t.item)); } - // Best item ranks above the worst item. assert!(top[0].score >= bottom[bottom.len() - 1].score); } #[test] fn top_bottom_small_group_has_empty_bottom() { - let mut g = mk_group(); - apply(&mut g, vote(1, "a", "b", 2, 1)); - let (top, bottom) = top_bottom(&g, 5); + let mut scope = mk_scope(); + apply(&mut scope, vote(1, "a", "b", 2, 1)); + let (top, bottom) = top_bottom(&scope, 5); assert_eq!(top.len(), 2); assert!(bottom.is_empty()); } #[test] fn connected_components_split_disconnected_pairs() { - let mut g = mk_group(); - // Two disconnected edges: (a,b) and (c,d) - apply(&mut g, vote(1, "a", "b", 3, 1)); - apply(&mut g, vote(2, "c", "d", 3, 1)); - - let n = g.idx_to_item.len(); - let (mut comps, isolates) = - connected_components_from_voted_pairs(n, g.voted_pairs.iter().copied()); + let mut scope = mk_scope(); + apply(&mut scope, vote(1, "a", "b", 3, 1)); + apply(&mut scope, vote(2, "c", "d", 3, 1)); + + let (_, idx_to_item) = item_index(&scope); + let (mut comps, isolates, _) = scope_components(&scope); assert!(isolates.is_empty()); - // Order-independent: sort components by their item names for stable assert. comps.sort_by_key(|c| { c.iter() - .map(|&i| g.idx_to_item[i].clone()) + .map(|&i| idx_to_item[i].clone()) .collect::>() }); assert_eq!(comps.len(), 2); - let comp0 = comps[0] - .iter() - .map(|&i| g.idx_to_item[i].as_str()) - .collect::>(); - let comp1 = comps[1] - .iter() - .map(|&i| g.idx_to_item[i].as_str()) - .collect::>(); - assert_eq!(comp0, vec!["a", "b"]); - assert_eq!(comp1, vec!["c", "d"]); } - /// A random spanning tree over 26 items needs only n−1 = 25 pairwise votes. - /// When each vote uses the "perfect" ratio (strength left : strength right = - /// (idx_left+1) : (idx_right+1)), rank centrality recovers the true order. - /// See `rank-eric.py` (Eric's demo of Negahban–Oh–Shah rank centrality). #[test] fn twenty_five_random_votes_perfect_ratios_sort_alphabet() { use rand::seq::SliceRandom; @@ -390,13 +403,13 @@ mod tests { let mut perm: Vec = (0..N).collect(); perm.shuffle(&mut rng); - let mut g = mk_group(); + let mut scope = mk_scope(); for k in 1..N { let i = *perm[..k].choose(&mut rng).unwrap(); let j = perm[k]; let (a, b) = (letters[i], letters[j]); apply( - &mut g, + &mut scope, vote( k as i64, &a.to_string(), @@ -407,34 +420,25 @@ mod tests { ); } - let ranked = ranked_items(&g); + let ranked = ranked_items(&scope); assert_eq!(ranked.len(), N); for (rank, item) in ranked.iter().enumerate() { let expected = char::from(b'a' + (N - 1 - rank) as u8); - assert_eq!( - item.item.as_str(), - expected.to_string(), - "rank {rank}: expected '{expected}', got '{}'", - item.item.as_str() - ); + assert_eq!(item.item.as_str(), expected.to_string()); } } #[test] fn subset_ranking_ranks_within_component_only() { - let mut g = mk_group(); - apply(&mut g, vote(1, "a", "b", 3, 1)); // a > b - apply(&mut g, vote(2, "c", "d", 1, 4)); // d > c - - let (comps, _) = connected_components_from_voted_pairs( - g.idx_to_item.len(), - g.voted_pairs.iter().copied(), - ); + let mut scope = mk_scope(); + apply(&mut scope, vote(1, "a", "b", 3, 1)); + apply(&mut scope, vote(2, "c", "d", 1, 4)); + + let (comps, _, _) = scope_components(&scope); assert_eq!(comps.len(), 2); - // Rank each component and ensure winner is first within that component. for comp in comps { - let ranked = ranked_items_subset(&g, &comp, 10000, 1e-8); + let ranked = ranked_items_subset(&scope, &comp, 10000, 1e-8); assert_eq!(ranked.len(), 2); let names = ranked.iter().map(|r| r.item.as_str()).collect::>(); if names.contains(&"a") { diff --git a/server/src/reducer.rs b/server/src/reducer.rs index 8c4c9f83635cbcbb037d680dbae2aa99cb2dc785..e6d8c8d2fe763f4928cfbd7c300c8d9768968a3f 100644 --- a/server/src/reducer.rs +++ b/server/src/reducer.rs @@ -4,6 +4,24 @@ use serde::{Deserialize, Serialize}; use crate::path_types::ItemId; +/// `(actor_uuid, min_item_id, max_item_id)` — one vote slot per human per pair. +pub type UuidVoteKey = (String, String, String); + +pub fn canonical_pair_ids(a: &ItemId, b: &ItemId) -> (String, String) { + let ak = a.as_str().to_string(); + let bk = b.as_str().to_string(); + if ak <= bk { + (ak, bk) + } else { + (bk, ak) + } +} + +pub fn uuid_vote_key(actor_uuid: &str, a: &ItemId, b: &ItemId) -> UuidVoteKey { + let (lo, hi) = canonical_pair_ids(a, b); + (actor_uuid.to_string(), lo, hi) +} + /// Parsed pairwise vote (internal representation). #[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] pub struct VoteData { @@ -44,118 +62,20 @@ impl VoteData { } } +/// Votes cast within one ranking scope (parent node). Edges and rankings are +/// derived on demand from [`Self::uuid_votes`]. #[derive(Debug, Clone, Default, Serialize, Deserialize)] -pub struct GroupState { - pub item_to_idx: HashMap, - pub idx_to_item: Vec, - pub edges: HashMap<(usize, usize), f64>, - pub voted_pairs: HashSet<(usize, usize)>, - /// Latest vote per `(actor_uuid, min_idx, max_idx)` — Sybil dedup anchor. - pub uuid_votes: HashMap<(String, usize, usize), VoteData>, +pub struct ScopeVotes { + pub uuid_votes: HashMap, pub recent_votes: Vec, } -impl GroupState { - pub fn new() -> Self { - Self { - item_to_idx: HashMap::new(), - idx_to_item: Vec::new(), - edges: HashMap::new(), - voted_pairs: HashSet::new(), - uuid_votes: HashMap::new(), - recent_votes: Vec::new(), - } - } - - fn ensure_item(&mut self, item: &ItemId) -> usize { - if let Some(&idx) = self.item_to_idx.get(item) { - return idx; - } - let idx = self.idx_to_item.len(); - self.idx_to_item.push(item.clone()); - self.item_to_idx.insert(item.clone(), idx); - idx - } - - fn add_edge_weight(&mut self, src: usize, dst: usize, w: f64) { - if w <= 0.0 { - return; - } - *self.edges.entry((src, dst)).or_insert(0.0) += w; - } - - fn subtract_edge_weight(&mut self, src: usize, dst: usize, w: f64) { - if w <= 0.0 { - return; - } - if let Some(entry) = self.edges.get_mut(&(src, dst)) { - *entry -= w; - if *entry <= 0.0 { - self.edges.remove(&(src, dst)); - } - } - } - - fn apply_weights(&mut self, vote: &VoteData, a_idx: usize, b_idx: usize) { - let w_a = vote.ratio_left as f64 * vote.trust_weight; - let w_b = vote.ratio_right as f64 * vote.trust_weight; - let (i, j) = if a_idx < b_idx { - (a_idx, b_idx) - } else { - (b_idx, a_idx) - }; - self.voted_pairs.insert((i, j)); - self.add_edge_weight(b_idx, a_idx, w_a); - self.add_edge_weight(a_idx, b_idx, w_b); - } - - fn rollback_weights(&mut self, vote: &VoteData) { - let a_idx = match self.item_to_idx.get(&vote.a) { - Some(&i) => i, - None => return, - }; - let b_idx = match self.item_to_idx.get(&vote.b) { - Some(&i) => i, - None => return, - }; - let w_a = vote.ratio_left as f64 * vote.trust_weight; - let w_b = vote.ratio_right as f64 * vote.trust_weight; - self.subtract_edge_weight(b_idx, a_idx, w_a); - self.subtract_edge_weight(a_idx, b_idx, w_b); - } - - /// Apply a validated vote, deduplicating by `actor_uuid` per unordered pair. +impl ScopeVotes { pub fn apply_vote(&mut self, vote: VoteData, actor_uuid: &str) { - let a_idx = self.ensure_item(&vote.a); - let b_idx = self.ensure_item(&vote.b); - let (i, j) = if a_idx < b_idx { - (a_idx, b_idx) - } else { - (b_idx, a_idx) - }; - - let dedupe_key = (actor_uuid.to_string(), i, j); - if let Some(old) = self.uuid_votes.get(&dedupe_key).cloned() { - self.rollback_weights(&old); - } - - self.apply_weights(&vote, a_idx, b_idx); - self.uuid_votes.insert(dedupe_key, vote.clone()); + let key = uuid_vote_key(actor_uuid, &vote.a, &vote.b); + self.uuid_votes.insert(key, vote.clone()); self.recent_votes.push(vote); } - - /// Rebuild edge weights from deduped uuid votes (load path — no rollback). - pub fn ingest_uuid_vote(&mut self, vote: VoteData, actor_uuid: &str) { - let a_idx = self.ensure_item(&vote.a); - let b_idx = self.ensure_item(&vote.b); - let (i, j) = if a_idx < b_idx { - (a_idx, b_idx) - } else { - (b_idx, a_idx) - }; - self.apply_weights(&vote, a_idx, b_idx); - self.uuid_votes.insert((actor_uuid.to_string(), i, j), vote); - } } /// Structured data imported from Reddit or elsewhere. @@ -179,7 +99,7 @@ pub struct NodeState { /// Ephemeral display view (Reddit title/author/etc.; not event-logged). pub data: Option, pub children: HashSet, - pub local_ranking: GroupState, + pub votes: ScopeVotes, } impl NodeState { @@ -242,7 +162,7 @@ impl GlobalTree { if let Some(node) = self.nodes.get_mut(parent) { node.children.insert(vote.a.clone()); node.children.insert(vote.b.clone()); - node.local_ranking.apply_vote(vote, actor_uuid); + node.votes.apply_vote(vote, actor_uuid); } } @@ -276,6 +196,7 @@ impl GlobalTree { #[cfg(test)] mod tests { use super::*; + use crate::ranking::edge_weight_sum; fn vote(ts: i64, a: &str, b: &str, l: i32, r: i32, pseudonym: &str) -> VoteData { VoteData { @@ -310,26 +231,23 @@ mod tests { #[test] fn same_uuid_replaces_prior_vote_on_pair() { - let mut g = GroupState::new(); + let mut scope = ScopeVotes::default(); let uuid = "u1"; - g.apply_vote(vote(1, "a", "b", 2, 1, "alice"), uuid); - let first_total: f64 = g.edges.values().sum(); - assert_eq!(first_total, 3.0); + scope.apply_vote(vote(1, "a", "b", 2, 1, "alice"), uuid); + assert_eq!(edge_weight_sum(&scope), 3.0); - g.apply_vote(vote(2, "a", "b", 0, 1, "bob"), uuid); - let second_total: f64 = g.edges.values().sum(); - assert_eq!(second_total, 1.0); - assert_eq!(g.uuid_votes.len(), 1); + scope.apply_vote(vote(2, "a", "b", 0, 1, "bob"), uuid); + assert_eq!(edge_weight_sum(&scope), 1.0); + assert_eq!(scope.uuid_votes.len(), 1); } #[test] fn different_uuids_both_count() { - let mut g = GroupState::new(); - g.apply_vote(vote(1, "a", "b", 2, 1, "alice"), "u1"); - g.apply_vote(vote(2, "a", "b", 0, 1, "bob"), "u2"); - let total: f64 = g.edges.values().sum(); - assert_eq!(total, 4.0); - assert_eq!(g.uuid_votes.len(), 2); + let mut scope = ScopeVotes::default(); + scope.apply_vote(vote(1, "a", "b", 2, 1, "alice"), "u1"); + scope.apply_vote(vote(2, "a", "b", 0, 1, "bob"), "u2"); + assert_eq!(edge_weight_sum(&scope), 4.0); + assert_eq!(scope.uuid_votes.len(), 2); } #[test] diff --git a/server/src/state.rs b/server/src/state.rs index 8dabc93e95c39cbe18d69596dee79683b384ef86..dcb82ff4beeaac8f0820de0e0131ca1f6bd81dcc 100644 --- a/server/src/state.rs +++ b/server/src/state.rs @@ -269,7 +269,7 @@ mod tests { use super::{normalize_scope, parse_item_param, AppConfig, AppState}; use crate::{ event_log::EventLog, events::Event, path_types::ItemId, projection_apply, - projection_store::ProjectionStore, reducer::EntityData, + projection_store::ProjectionStore, ranking::edge_weight_sum, reducer::EntityData, }; fn event_record(seq: u64, event: Event) -> crate::events::EventRecord { @@ -442,7 +442,7 @@ mod tests { assert_eq!(projection_store.last_applied_event_count().unwrap(), 1); let first = projection_store.scope_tree(&ItemId::root()).unwrap(); let first_root = first.get(&ItemId::root()).unwrap(); - let first_edge_total: f64 = first_root.local_ranking.edges.values().sum(); + let first_edge_total = edge_weight_sum(&first_root.votes); assert_eq!(first_edge_total, 3.0); super::catch_up_projection(&log, &projection_store) @@ -451,7 +451,7 @@ mod tests { assert_eq!(projection_store.last_applied_event_count().unwrap(), 1); let second = projection_store.scope_tree(&ItemId::root()).unwrap(); let second_root = second.get(&ItemId::root()).unwrap(); - let second_edge_total: f64 = second_root.local_ranking.edges.values().sum(); + let second_edge_total = edge_weight_sum(&second_root.votes); assert_eq!(second_edge_total, first_edge_total); } @@ -529,9 +529,8 @@ mod tests { let root = projected.get(&ItemId::root()).unwrap(); assert!(root.children.contains(&ItemId::parse("alpha").unwrap())); assert!(root.children.contains(&ItemId::parse("beta").unwrap())); - assert_eq!(root.local_ranking.idx_to_item.len(), 2); - let edge_total: f64 = root.local_ranking.edges.values().sum(); - assert_eq!(edge_total, 3.0); + assert_eq!(crate::ranking::ranked_items(&root.votes).len(), 2); + assert_eq!(edge_weight_sum(&root.votes), 3.0); } #[tokio::test] @@ -620,7 +619,7 @@ mod tests { let root = tree.get(&ItemId::root()).unwrap(); assert!(root.children.contains(&ItemId::parse("beta").unwrap())); assert!(root.children.contains(&ItemId::parse("gamma").unwrap())); - assert_eq!(root.local_ranking.idx_to_item.len(), 3); + assert_eq!(crate::ranking::ranked_items(&root.votes).len(), 3); } #[test] diff --git a/server/src/storage_schema.rs b/server/src/storage_schema.rs index 67c9f4ab2e1865f8da81b6735dd2a05b87e5e366..fe2671b876f3df77fdfd402dbc845ff1eb3512cd 100644 --- a/server/src/storage_schema.rs +++ b/server/src/storage_schema.rs @@ -3,86 +3,53 @@ //! //! Votes are stored as deduped `uuid_votes` entries plus an append-only //! `recent_votes` audit list. Edge weights for rank centrality are derived -//! from `uuid_votes` on read, not incrementally merged in RocksDB. +//! from `uuid_votes` on read, not stored in RocksDB. -use std::collections::{HashSet}; +use std::collections::HashSet; use durable::{Batch, Db, Durable, Leaf, List, Map}; use crate::{ path_types::ItemId, - reducer::{EntityData, GroupState, NodeState, VoteData}, + reducer::{EntityData, NodeState, ScopeVotes, VoteData, UuidVoteKey, uuid_vote_key}, storage_dto::{ decode_entity_data, decode_vote, encode_entity_data, encode_vote, parse_stored_id, StoredEntityDataV1, StoredVoteV1, }, }; -/// `(actor_uuid, min_item_id, max_item_id)` — one vote slot per human per pair. -pub type UuidVoteKey = (String, String, String); - /// One node in the fractal tree, exploded into precisely-updatable collections. #[derive(Durable)] #[allow(dead_code)] pub struct NodeSchema { - /// Presence marker (a node "exists" once ensured/voted/imported). pub present: Leaf, - /// Domain-specific derived view (Reddit title/author/…); absent => None. pub data: Leaf, - /// Child ids (a set; value is always `true`). pub children: Map>, - /// Latest vote per actor per unordered pair; edges are derived from this on read. pub uuid_votes: Map>, - /// Recent votes, append-only oldest-first (cap applied on read). pub recent_votes: List>, - /// When ephemeral Reddit display content was last fetched (ms); absent after eviction. pub fetched_at: Leaf, } -/// The single database root: nodes, identity maps, view counts, and metadata. #[derive(Durable)] #[allow(dead_code)] pub struct Store { pub nodes: Map, - /// Global pseudonym → actor UUID (Sybil dedup anchor). pub pseudonyms: Map>, pub proj_meta: Map>, pub view_counts: Map>, pub view_meta: Map>, } -/// Max recent votes returned when loading a node (query-time cap only). pub const RECENT_VOTES_CAP: u64 = 200; fn id_key(id: &ItemId) -> String { id.as_str().to_string() } -fn pair_keys(a: &ItemId, b: &ItemId) -> (String, String) { - let ak = id_key(a); - let bk = id_key(b); - if ak <= bk { - (ak, bk) - } else { - (bk, ak) - } -} - -pub fn uuid_vote_key(actor_uuid: &str, a: &ItemId, b: &ItemId) -> UuidVoteKey { - let (lo, hi) = pair_keys(a, b); - (actor_uuid.to_string(), lo, hi) -} - -/// Path to a node by id. pub fn node(id: &ItemId) -> durable::Path { Store::root().nodes().key(&id_key(id)) } -// --------------------------------------------------------------------------- -// Reconstruction (durable -> in-memory) -// --------------------------------------------------------------------------- - -/// Reconstruct a node's in-memory state, or `None` if the node does not exist. pub fn load_node_state(db: &Db, id: &ItemId) -> durable::Result> { let np = node(id); let present = np.present().get(db)?.unwrap_or(false); @@ -100,47 +67,39 @@ pub fn load_node_state(db: &Db, id: &ItemId) -> durable::Result) -> durable::Result { - let mut group = GroupState::new(); +fn load_scope_votes(db: &Db, np: &durable::Path) -> durable::Result { + let mut votes = ScopeVotes::default(); for (key, stored) in np.uuid_votes().iter(db)? { - let (actor_uuid, _lo, _hi) = key; let vote = decode_vote(stored).map_err(durable::Error::Deserialize)?; - group.ingest_uuid_vote(vote, &actor_uuid); + votes.uuid_votes.insert(key, vote); } let stored = np.recent_votes().iter(db)?; let cap = RECENT_VOTES_CAP as usize; let start = stored.len().saturating_sub(cap); - group.recent_votes = stored[start..] + votes.recent_votes = stored[start..] .iter() .map(|s| decode_vote(s.clone()).map_err(durable::Error::Deserialize)) .collect::, _>>()?; - Ok(group) + Ok(votes) } fn parse_storage_id(s: &str) -> durable::Result { parse_stored_id(s).map_err(durable::Error::Deserialize) } -// --------------------------------------------------------------------------- -// Write helpers (event -> reified point updates on a batch) -// --------------------------------------------------------------------------- - -/// Wire a node and its ancestors into the tree exactly like -/// [`crate::reducer::GlobalTree::ensure_path`]: set presence and parent→child -/// links along the canonical breadcrumb path. pub fn ensure_path_writes(batch: &mut Batch, id: &ItemId) { let root = ItemId::root(); batch.write(node(&root).present().set(&true)); @@ -161,7 +120,6 @@ pub fn ensure_path_writes(batch: &mut Batch, id: &ItemId) { } } -/// Reified writes for a validated vote under `parent`. pub fn vote_writes( batch: &mut Batch, parent: &ItemId, @@ -182,14 +140,12 @@ pub fn vote_writes( Ok(()) } -/// Reified writes for ephemeral Reddit display content (not event-logged). pub fn entity_content_writes(batch: &mut Batch, id: &ItemId, view: &EntityData, fetched_at: i64) { ensure_path_writes(batch, id); batch.write(node(id).data().set(&encode_entity_data(view))); batch.write(node(id).fetched_at().set(&fetched_at)); } -/// Clear cached display content for one node (structure/votes are untouched). pub fn entity_content_clear_writes(batch: &mut Batch, id: &ItemId) { batch.write(node(id).data().delete()); batch.write(node(id).fetched_at().delete()); @@ -199,6 +155,7 @@ pub fn entity_content_clear_writes(batch: &mut Batch, id: &ItemId) { mod tests { use super::*; use crate::identity::{seed_default_pseudonym, DEFAULT_ACTOR_UUID, DEFAULT_PSEUDONYM}; + use crate::ranking::edge_weight_sum; fn sample_vote(ts: i64, a: &str, b: &str, l: i32, r: i32) -> VoteData { VoteData { @@ -213,7 +170,7 @@ mod tests { } #[test] - fn vote_roundtrip_reconstructs_group_state() { + fn vote_roundtrip_reconstructs_ranking() { let dir = tempfile::tempdir().unwrap(); let db = Db::open(dir.path()).unwrap(); seed_default_pseudonym(&db).unwrap(); @@ -225,11 +182,8 @@ mod tests { batch.commit().unwrap(); let node_state = load_node_state(&db, &parent).unwrap().unwrap(); - let g = &node_state.local_ranking; - assert_eq!(g.idx_to_item.len(), 2); - let edge_total: f64 = g.edges.values().sum(); - assert_eq!(edge_total, 3.0); - assert_eq!(g.recent_votes.len(), 1); + assert_eq!(edge_weight_sum(&node_state.votes), 3.0); + assert_eq!(node_state.votes.recent_votes.len(), 1); assert!(node_state.children.contains(&ItemId::opaque("alpha"))); assert!(node_state.children.contains(&ItemId::opaque("beta"))); } @@ -263,10 +217,9 @@ mod tests { .unwrap(); batch.commit().unwrap(); - let g = &load_node_state(&db, &parent).unwrap().unwrap().local_ranking; - let edge_total: f64 = g.edges.values().sum(); - assert_eq!(edge_total, 1.0); - assert_eq!(g.uuid_votes.len(), 1); + let votes = &load_node_state(&db, &parent).unwrap().unwrap().votes; + assert_eq!(edge_weight_sum(votes), 1.0); + assert_eq!(votes.uuid_votes.len(), 1); } #[test] @@ -293,15 +246,8 @@ mod tests { ); let node_state = load_node_state(&db, &parent).unwrap().unwrap(); - assert_eq!(node_state.local_ranking.recent_votes.len(), RECENT_VOTES_CAP as usize); - assert_eq!( - node_state - .local_ranking - .recent_votes - .first() - .map(|v| v.ts), - Some(10) - ); + assert_eq!(node_state.votes.recent_votes.len(), RECENT_VOTES_CAP as usize); + assert_eq!(node_state.votes.recent_votes.first().map(|v| v.ts), Some(10)); } #[test] diff --git a/server/tests/integration_ui.rs b/server/tests/integration_ui.rs index cc7a16d756673b95ba336f2d6130eaf40908cc60..40e4cfe1eec36dad29a075fb01aefb4dffd856d0 100644 --- a/server/tests/integration_ui.rs +++ b/server/tests/integration_ui.rs @@ -132,7 +132,7 @@ async fn post_ui_record_vote_morphs_ranking_and_persists() { let state = create_app_state(cfg).await; let tree = state.scope_tree(&ItemId::root()).unwrap(); let root = tree.get(&ItemId::root()).expect("root node after replay"); - let ranked = sorter2_server::ranking::ranked_items(&root.local_ranking); + let ranked = sorter2_server::ranking::ranked_items(&root.votes); assert_eq!(ranked.len(), 2); assert_eq!(ranked[0].item.as_str(), "alpha"); } Side B — contributor: tommy-mor Side B — commit message: [239c074b] url schema stuff Side B — unified diff (full patch): diff --git a/AGENTS.md b/AGENTS.md index 426a88e7c1da54fe0a28c5c76fa4e1f1bc117fcf..e60b9ba6012593361ef10e8fdd9439cd9932e09b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -58,3 +58,4 @@ Use **tmux** for `cargo run --package sorter2-server` (dev server). Rebuild afte - First `cargo test` / `cargo build --release` is slow; Clojure smoke test always does a release build. - `legacy/` and `ideas/` are not part of the workspace build. +- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/`](server/src/url_rules/) (composable Rust, not a config DSL). After changing canonicalization rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`. diff --git a/Cargo.lock b/Cargo.lock index 0dd4fce5fb6400ae153cca4e3dbf5a5158e6d8b4..49a908ef935c430dbe63c6a28d8a24e38b489486 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1951,6 +1951,7 @@ dependencies = [ "tower-http 0.5.2", "tracing", "tracing-subscriber", + "url", "urlencoding", ] diff --git a/REPLAY.sh b/REPLAY.sh new file mode 100755 index 0000000000000000000000000000000000000000..f2dbd8aea60c02d2feef74805f7ef5c2b7022537 --- /dev/null +++ b/REPLAY.sh @@ -0,0 +1,2 @@ +cargo run --package sorter2-server -- replay-index + diff --git a/server/Cargo.toml b/server/Cargo.toml index 27f552c20b97ef28cdde4cb6b1a4980375135111..ad4912791aff59fb1d3293f66ad381ae618cd60b 100644 --- a/server/Cargo.toml +++ b/server/Cargo.toml @@ -24,6 +24,7 @@ async-stream = "0.3" futures-util = { version = "0.3", default-features = false, features = ["std"] } rand = "0.8" urlencoding = "2" +url = "2" durable = { path = "../durable" } [dev-dependencies] diff --git a/server/src/entity_store.rs b/server/src/entity_store.rs index d5d17c3676e4a8ddec998e9f5a9dbafe9c2d9d0e..d29f39aecca6f12cdcf263cf77c3654eb4ee6cfa 100644 --- a/server/src/entity_store.rs +++ b/server/src/entity_store.rs @@ -124,7 +124,7 @@ mod tests { fn round_trip_payload() { let tmp = tempfile::tempdir().unwrap(); let store = EntityStore::open(tmp.path()).unwrap(); - let id = ItemId::parse("reddit.com/r/rust").unwrap(); + let id = ItemId::from_url("https://reddit.com/r/rust").unwrap(); let payload = json!({"kind": "t5", "data": {"display_name": "rust"}}); store.put(&id, &payload).unwrap(); diff --git a/server/src/event_log.rs b/server/src/event_log.rs index 36f5b406084065b608735987cdb483c236e03081..2c9290b6fdbf2c2ad1c0f1ffd7374b2d9cc97f36 100644 --- a/server/src/event_log.rs +++ b/server/src/event_log.rs @@ -199,7 +199,7 @@ mod tests { log.append(&sample_record( 1, Event::NodeEnsured { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), }, )) .await @@ -237,7 +237,7 @@ mod tests { let path = tmp.path().join("events.jsonl"); let log = EventLog::new(&path); let event = Event::NodeEnsured { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), }; log.append(&sample_record(1, event)).await.unwrap(); @@ -255,7 +255,7 @@ mod tests { let path = tmp.path().join("events.jsonl"); std::fs::write( &path, - r#"{"type":"node_ensured","id":"reddit.com/r/rust"} + r#"{"type":"node_ensured","id":"https://reddit.com/r/rust"} {"schema":1,"seq":1,"ts":1,"event":{"type":"vote_recorded","ts":1,"a":"a","b":"b","ratio_left":2,"ratio_right":1,"scope":""}} "#, ) @@ -295,7 +295,7 @@ mod tests { log.append(&sample_record( 1, Event::NodeEnsured { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), }, )) .await @@ -303,7 +303,7 @@ mod tests { log.append(&sample_record( 3, Event::NodeEnsured { - id: "reddit.com/r/python".into(), + id: "https://reddit.com/r/python".into(), }, )) .await diff --git a/server/src/journal.rs b/server/src/journal.rs index 521a108019de1ea870d14c4fafbfe572c20ce0de..50bc89f976edb82b7b0e49e954a8eccbbe82bf87 100644 --- a/server/src/journal.rs +++ b/server/src/journal.rs @@ -141,10 +141,10 @@ mod tests { let j2 = journal.clone(); let (r1, r2) = tokio::join!( j1.append(Event::NodeEnsured { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), }), j2.append(Event::NodeEnsured { - id: "reddit.com/r/python".into(), + id: "https://reddit.com/r/python".into(), }), ); r1.unwrap(); @@ -153,10 +153,10 @@ mod tests { assert_eq!(projection_store.last_applied_event_count().unwrap(), 2); let tree = projection_store.load_tree().unwrap(); assert!(tree - .get(&ItemId::parse("reddit.com/r/rust").unwrap()) + .get(&ItemId::parse("https://reddit.com/r/rust").unwrap()) .is_some()); assert!(tree - .get(&ItemId::parse("reddit.com/r/python").unwrap()) + .get(&ItemId::parse("https://reddit.com/r/python").unwrap()) .is_some()); } @@ -170,7 +170,7 @@ mod tests { 1, 1, Event::NodeEnsured { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), }, )) .await @@ -186,7 +186,7 @@ mod tests { 1, 1, Event::NodeEnsured { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), }, )], ) @@ -202,7 +202,7 @@ mod tests { ); journal .append(Event::NodeEnsured { - id: "reddit.com/r/python".into(), + id: "https://reddit.com/r/python".into(), }) .await .unwrap(); @@ -227,13 +227,13 @@ mod tests { journal .append_many(vec![ Event::NodeEnsured { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), }, Event::NodeEnsured { - id: "reddit.com/r/python".into(), + id: "https://reddit.com/r/python".into(), }, Event::NodeEnsured { - id: "reddit.com/r/clojure".into(), + id: "https://reddit.com/r/clojure".into(), }, ]) .await @@ -245,7 +245,7 @@ mod tests { assert_eq!(projection_store.last_applied_event_count().unwrap(), 3); let tree = projection_store.load_tree().unwrap(); assert!(tree - .get(&ItemId::parse("reddit.com/r/clojure").unwrap()) + .get(&ItemId::parse("https://reddit.com/r/clojure").unwrap()) .is_some()); } } diff --git a/server/src/lib.rs b/server/src/lib.rs index 9bd5f76fd1406b9b1be4c272f4ba8647edde2678..5c02c8e704e4664453bad75d819df8a067668176 100644 --- a/server/src/lib.rs +++ b/server/src/lib.rs @@ -9,6 +9,7 @@ pub mod journal; pub mod pair; pub mod parser; pub mod path_types; +pub mod url_rules; pub mod projection_apply; pub mod projection_store; pub mod ranking; diff --git a/server/src/pair.rs b/server/src/pair.rs index 43f780ba6ea6ce1cdc2e1f4cbb252ba8a10684b9..815a97b80e3e9f348e0937a4f147f2862018edb0 100644 --- a/server/src/pair.rs +++ b/server/src/pair.rs @@ -381,42 +381,42 @@ mod tests { #[test] fn suggest_prefers_unvoted_pair() { - let parent = ItemId::parse("reddit.com/r/rust").unwrap(); + let parent = ItemId::parse("https://reddit.com/r/rust").unwrap(); let mut tree = seed_children( &parent, &[ - "reddit.com/r/rust/a", - "reddit.com/r/rust/b", - "reddit.com/r/rust/c", + "https://reddit.com/r/rust/a", + "https://reddit.com/r/rust/b", + "https://reddit.com/r/rust/c", ], ); let vote = - VoteData::from_recorded(1, "reddit.com/r/rust/a", "reddit.com/r/rust/b", 2, 1).unwrap(); + VoteData::from_recorded(1, "https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b", 2, 1).unwrap(); tree.apply_vote(&parent, vote); let group = tree.get(&parent).unwrap().local_ranking.clone(); let pool = children_of(&tree, &parent); let (l, r) = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); - let voted_ab = (l.as_str() == "reddit.com/r/rust/a" && r.as_str() == "reddit.com/r/rust/b") - || (l.as_str() == "reddit.com/r/rust/b" && r.as_str() == "reddit.com/r/rust/a"); + let voted_ab = (l.as_str() == "https://reddit.com/r/rust/a" && r.as_str() == "https://reddit.com/r/rust/b") + || (l.as_str() == "https://reddit.com/r/rust/b" && r.as_str() == "https://reddit.com/r/rust/a"); assert!(!voted_ab); } #[test] fn suggest_bridges_separate_components() { - let parent = ItemId::parse("reddit.com/r/rust").unwrap(); + let parent = ItemId::parse("https://reddit.com/r/rust").unwrap(); let mut tree = seed_children( &parent, &[ - "reddit.com/r/rust/a", - "reddit.com/r/rust/b", - "reddit.com/r/rust/c", - "reddit.com/r/rust/d", + "https://reddit.com/r/rust/a", + "https://reddit.com/r/rust/b", + "https://reddit.com/r/rust/c", + "https://reddit.com/r/rust/d", ], ); let ab = - VoteData::from_recorded(1, "reddit.com/r/rust/a", "reddit.com/r/rust/b", 2, 1).unwrap(); + VoteData::from_recorded(1, "https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b", 2, 1).unwrap(); let cd = - VoteData::from_recorded(2, "reddit.com/r/rust/c", "reddit.com/r/rust/d", 2, 1).unwrap(); + VoteData::from_recorded(2, "https://reddit.com/r/rust/c", "https://reddit.com/r/rust/d", 2, 1).unwrap(); tree.apply_vote(&parent, ab); tree.apply_vote(&parent, cd); let group = tree.get(&parent).unwrap().local_ranking.clone(); @@ -424,37 +424,37 @@ mod tests { let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); let from_ab = - chosen.contains("reddit.com/r/rust/a") || chosen.contains("reddit.com/r/rust/b"); + chosen.contains("https://reddit.com/r/rust/a") || chosen.contains("https://reddit.com/r/rust/b"); let from_cd = - chosen.contains("reddit.com/r/rust/c") || chosen.contains("reddit.com/r/rust/d"); + chosen.contains("https://reddit.com/r/rust/c") || chosen.contains("https://reddit.com/r/rust/d"); assert!(from_ab && from_cd, "expected bridge pair, got {:?}", chosen); } #[test] fn suggest_prefers_attach_over_isolate_pair_among_many_unranked() { - let parent = ItemId::parse("reddit.com/r/rust").unwrap(); + let parent = ItemId::parse("https://reddit.com/r/rust").unwrap(); let mut tree = seed_children( &parent, &[ - "reddit.com/r/rust/a", - "reddit.com/r/rust/b", - "reddit.com/r/rust/c", - "reddit.com/r/rust/d", - "reddit.com/r/rust/e", + "https://reddit.com/r/rust/a", + "https://reddit.com/r/rust/b", + "https://reddit.com/r/rust/c", + "https://reddit.com/r/rust/d", + "https://reddit.com/r/rust/e", ], ); let ab = - VoteData::from_recorded(1, "reddit.com/r/rust/a", "reddit.com/r/rust/b", 2, 1).unwrap(); + VoteData::from_recorded(1, "https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b", 2, 1).unwrap(); tree.apply_vote(&parent, ab); let group = tree.get(&parent).unwrap().local_ranking.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); let from_ab = - chosen.contains("reddit.com/r/rust/a") || chosen.contains("reddit.com/r/rust/b"); - let from_cde = chosen.contains("reddit.com/r/rust/c") - || chosen.contains("reddit.com/r/rust/d") - || chosen.contains("reddit.com/r/rust/e"); + chosen.contains("https://reddit.com/r/rust/a") || chosen.contains("https://reddit.com/r/rust/b"); + let from_cde = chosen.contains("https://reddit.com/r/rust/c") + || chosen.contains("https://reddit.com/r/rust/d") + || chosen.contains("https://reddit.com/r/rust/e"); assert!( from_ab && from_cde, "expected ranked+unranked attach, got {:?}", @@ -464,40 +464,40 @@ mod tests { #[test] fn suggest_connects_isolate_to_existing_component() { - let parent = ItemId::parse("reddit.com/r/rust").unwrap(); + let parent = ItemId::parse("https://reddit.com/r/rust").unwrap(); let mut tree = seed_children( &parent, &[ - "reddit.com/r/rust/a", - "reddit.com/r/rust/b", - "reddit.com/r/rust/c", + "https://reddit.com/r/rust/a", + "https://reddit.com/r/rust/b", + "https://reddit.com/r/rust/c", ], ); let ab = - VoteData::from_recorded(1, "reddit.com/r/rust/a", "reddit.com/r/rust/b", 2, 1).unwrap(); + VoteData::from_recorded(1, "https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b", 2, 1).unwrap(); tree.apply_vote(&parent, ab); let group = tree.get(&parent).unwrap().local_ranking.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); - assert!(chosen.contains("reddit.com/r/rust/c")); - assert!(chosen.contains("reddit.com/r/rust/a") || chosen.contains("reddit.com/r/rust/b")); + assert!(chosen.contains("https://reddit.com/r/rust/c")); + assert!(chosen.contains("https://reddit.com/r/rust/a") || chosen.contains("https://reddit.com/r/rust/b")); } #[test] fn suggest_zips_adjacent_ranks_when_tree_complete() { - let parent = ItemId::parse("reddit.com/r/rust").unwrap(); + let parent = ItemId::parse("https://reddit.com/r/rust").unwrap(); let mut tree = seed_children( &parent, &[ - "reddit.com/r/rust/a", - "reddit.com/r/rust/b", - "reddit.com/r/rust/c", + "https://reddit.com/r/rust/a", + "https://reddit.com/r/rust/b", + "https://reddit.com/r/rust/c", ], ); for (a, b, l, r) in [ - ("reddit.com/r/rust/a", "reddit.com/r/rust/b", 3, 1), - ("reddit.com/r/rust/a", "reddit.com/r/rust/c", 2, 1), + ("https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b", 3, 1), + ("https://reddit.com/r/rust/a", "https://reddit.com/r/rust/c", 2, 1), ] { let v = VoteData::from_recorded(1, a, b, l, r).unwrap(); tree.apply_vote(&parent, v); @@ -506,26 +506,26 @@ mod tests { let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); - assert!(chosen.contains("reddit.com/r/rust/b")); - assert!(chosen.contains("reddit.com/r/rust/c")); + assert!(chosen.contains("https://reddit.com/r/rust/b")); + assert!(chosen.contains("https://reddit.com/r/rust/c")); } #[test] fn suggest_zip_prefers_1v2_before_2v3_when_both_unvoted() { - let parent = ItemId::parse("reddit.com/r/rust").unwrap(); + let parent = ItemId::parse("https://reddit.com/r/rust").unwrap(); let mut tree = seed_children( &parent, &[ - "reddit.com/r/rust/a", - "reddit.com/r/rust/b", - "reddit.com/r/rust/c", - "reddit.com/r/rust/d", + "https://reddit.com/r/rust/a", + "https://reddit.com/r/rust/b", + "https://reddit.com/r/rust/c", + "https://reddit.com/r/rust/d", ], ); for (a, b, l, r) in [ - ("reddit.com/r/rust/c", "reddit.com/r/rust/d", 3, 1), - ("reddit.com/r/rust/b", "reddit.com/r/rust/c", 2, 1), - ("reddit.com/r/rust/a", "reddit.com/r/rust/c", 2, 1), + ("https://reddit.com/r/rust/c", "https://reddit.com/r/rust/d", 3, 1), + ("https://reddit.com/r/rust/b", "https://reddit.com/r/rust/c", 2, 1), + ("https://reddit.com/r/rust/a", "https://reddit.com/r/rust/c", 2, 1), ] { let v = VoteData::from_recorded(1, a, b, l, r).unwrap(); tree.apply_vote(&parent, v); @@ -534,16 +534,16 @@ mod tests { let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); - assert!(chosen.contains("reddit.com/r/rust/a")); - assert!(chosen.contains("reddit.com/r/rust/b")); + assert!(chosen.contains("https://reddit.com/r/rust/a")); + assert!(chosen.contains("https://reddit.com/r/rust/b")); } #[test] fn resolve_pair_picks_from_pool() { - let parent = ItemId::parse("reddit.com/r/rust").unwrap(); - let tree = seed_children(&parent, &["reddit.com/r/rust/a", "reddit.com/r/rust/b"]); + let parent = ItemId::parse("https://reddit.com/r/rust").unwrap(); + let tree = seed_children(&parent, &["https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b"]); let pair = resolve_pair(&tree, &parent, None, None).unwrap(); - let pool: HashSet<_> = ["reddit.com/r/rust/a", "reddit.com/r/rust/b"] + let pool: HashSet<_> = ["https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b"] .into_iter() .collect(); assert!(pool.contains(pair.0.as_str())); diff --git a/server/src/parser.rs b/server/src/parser.rs index 9df2dcc9313f7fe250ce3b6aa167f6ec5d57951f..b2a963dd6cab415576c8d3a9a241966758565bb8 100644 --- a/server/src/parser.rs +++ b/server/src/parser.rs @@ -23,7 +23,7 @@ mod tests { fn parses_short_path() { assert_eq!( parse_reddit_url("r/rust").unwrap().as_str(), - "reddit.com/r/rust" + "https://reddit.com/r/rust" ); } @@ -33,7 +33,7 @@ mod tests { parse_reddit_url("https://www.reddit.com/r/programming/hot") .unwrap() .as_str(), - "reddit.com/r/programming" + "https://reddit.com/r/programming" ); } @@ -43,7 +43,10 @@ mod tests { "https://old.reddit.com/r/AmItheAsshole/comments/1trnvdl/aita_for_cancelling/", ) .unwrap(); - assert_eq!(id.as_str(), "reddit.com/r/amitheasshole/comments/1trnvdl"); + assert_eq!( + id.as_str(), + "https://reddit.com/r/amitheasshole/comments/1trnvdl" + ); } #[test] diff --git a/server/src/path_types.rs b/server/src/path_types.rs index fafd924452f6fd85e7a5b27ed2653a19581e6e56..71ffc01f35c5589686ff05dae3b5610fa01f30ce 100644 --- a/server/src/path_types.rs +++ b/server/src/path_types.rs @@ -1,13 +1,14 @@ use serde::{Deserialize, Serialize}; use std::fmt; -/// Canonical hierarchical identity for any URL/path in the fractal tree. +use crate::url_rules::{looks_like_url, navigable_breadcrumbs, parent_url, resolve_canonical}; + +/// Canonical identity: a real URL (with scheme) or an opaque non-URL key. #[derive(Debug, Clone, Hash, PartialEq, Eq, PartialOrd, Ord, Serialize, Deserialize, Default)] pub struct ItemId(String); impl ItemId { - /// Parse an already-canonical path (no URL normalization). Empty string is invalid here; - /// use [`Self::root`] for the tree root. + /// Parse an already-canonical id (no normalization). Empty string is invalid; use [`Self::root`]. pub fn parse(s: &str) -> Option { let t = s.trim(); if t.is_empty() { @@ -16,12 +17,12 @@ impl ItemId { Some(Self(t.to_string())) } - /// Build an opaque item key (legacy demo votes, non-URL items). + /// Build an opaque item key (demo votes, non-URL items). pub fn opaque(s: impl Into) -> Self { Self(s.into()) } - /// Root of the internet tree (empty path). + /// Root of the internet tree. pub fn root() -> Self { Self(String::new()) } @@ -34,23 +35,18 @@ impl ItemId { &self.0 } - /// Creates a canonical ID from a raw URL or path. Normalizes domains and - /// trims tracking query params. + /// Canonical URL from a raw pasted or fetched URL. pub fn from_url(raw_url: &str) -> Option { - Self::canonicalize(raw_url).map(Self) + resolve_canonical(raw_url).map(Self) } - /// Normalize strings from forms, events, and Reddit imports into the same - /// stored id shape (e.g. drop post title slug after comment id). + /// Normalize strings from forms, events, and imports into canonical identity. pub fn from_storage(s: &str) -> Option { let t = s.trim(); if t.is_empty() { return None; } - if t.contains("://") || t.starts_with("r/") { - return Self::from_url(t).or_else(|| Self::parse(t)); - } - if t.starts_with("reddit.com/") && t.contains("/comments/") { + if looks_like_url(t) { return Self::from_url(t).or_else(|| Self::parse(t)); } Self::parse(t).or_else(|| Self::from_url(t)) @@ -62,35 +58,52 @@ impl ItemId { if s.is_empty() { return Self::root(); } - Self(format!("reddit.com/r/{s}")) + if looks_like_url(s) || s.contains('/') { + Self::from_storage(s).unwrap_or_else(|| Self::opaque(s)) + } else { + Self(format!("https://reddit.com/r/{s}")) + } } - /// Extract the parent, e.g. `reddit.com/r/aww/comments/1trnvdl` → - /// `reddit.com/r/aww`. + /// Immediate parent scope in the tree. pub fn parent(&self) -> Option { - if self.0.is_empty() { + if self.is_root() { return None; } - + if looks_like_url(self.0.as_str()) { + return parent_url(self.0.as_str()).map(Self); + } let parts: Vec<&str> = self.0.trim_end_matches('/').split('/').collect(); if parts.len() <= 1 { return None; } - - if self.0.contains("/comments/") { - return Some(Self(parts[..parts.len().saturating_sub(2)].join("/"))); - } - Some(Self(parts[..parts.len() - 1].join("/"))) } pub fn segments(&self) -> Vec<&str> { + if self.is_root() { + return vec![]; + } + if let Some(rest) = self.0.strip_prefix("https://") { + return rest.split('/').filter(|s| !s.is_empty()).collect(); + } + if let Some(rest) = self.0.strip_prefix("http://") { + return rest.split('/').filter(|s| !s.is_empty()).collect(); + } self.0.split('/').filter(|s| !s.is_empty()).collect() } - /// Cumulative paths for breadcrumb rendering, e.g. - /// `reddit.com/r/movies` → `["reddit.com", "reddit.com/r", "reddit.com/r/movies"]`. + /// Cumulative navigable paths for breadcrumbs and tree wiring (includes self). pub fn breadcrumb_paths(&self) -> Vec { + if self.is_root() { + return vec![]; + } + if looks_like_url(self.0.as_str()) { + return navigable_breadcrumbs(self.0.as_str()) + .into_iter() + .map(ItemId) + .collect(); + } let segs = self.segments(); let mut paths = Vec::with_capacity(segs.len()); let mut current = String::new(); @@ -111,13 +124,13 @@ impl ItemId { if self.is_root() { return String::new(); } - if self.as_str().contains("://") { - return self.as_str().to_string(); + if self.0.contains("://") { + return self.0.clone(); } if self.segments().first().is_some_and(|s| s.contains('.')) { - format!("https://{}", self.as_str()) + format!("https://{}", self.0) } else { - self.as_str().to_string() + self.0.clone() } } @@ -144,70 +157,6 @@ impl ItemId { pub fn from_browse_uri(path: &str) -> Option { path.strip_prefix("/~/").map(ItemId::from_browse_tail) } - - fn canonicalize(raw: &str) -> Option { - let s = raw.trim(); - if s.is_empty() { - return None; - } - - let owned = if let Some(rest) = s.strip_prefix("r/") { - format!("reddit.com/r/{rest}") - } else if let Some(rest) = s.strip_prefix("/r/") { - format!("reddit.com/r/{rest}") - } else { - s.to_string() - }; - - let (host_path, _query) = split_query(&owned); - let host_path = host_path.trim_end_matches('/'); - - let path = if host_path.contains("://") { - parse_url_host_path(host_path)? - } else if host_path.starts_with("reddit.com") || host_path.starts_with("www.reddit.com") { - normalize_reddit_host_path(host_path) - } else if host_path.contains('/') { - host_path.to_string() - } else { - return None; - }; - - Some(normalize_reddit_path(&path)) - } -} - -fn split_query(s: &str) -> (&str, Option<&str>) { - if let Some((path, q)) = s.split_once('?') { - (path, Some(q)) - } else { - (s, None) - } -} - -fn parse_url_host_path(url: &str) -> Option { - let rest = url - .strip_prefix("https://") - .or_else(|| url.strip_prefix("http://")) - .unwrap_or(url); - let (host, path) = rest.split_once('/').unwrap_or((rest, "")); - let host = normalize_host(host); - if path.is_empty() { - Some(host) - } else { - Some(format!("{host}/{path}")) - } -} - -fn normalize_host(host: &str) -> String { - let h = host - .strip_prefix("www.") - .unwrap_or(host) - .to_ascii_lowercase(); - if h == "old.reddit.com" || h == "new.reddit.com" || h == "reddit.com" { - "reddit.com".to_string() - } else { - h - } } fn normalize_browse_tail(tail: &str) -> String { @@ -215,7 +164,6 @@ fn normalize_browse_tail(tail: &str) -> String { if t.is_empty() { return String::new(); } - // Some HTTP stacks collapse `https://` → `https:/` inside a path segment. if t.starts_with("https:/") && !t.starts_with("https://") { return format!("https://{}", &t[7..]); } @@ -225,33 +173,6 @@ fn normalize_browse_tail(tail: &str) -> String { t.to_string() } -fn normalize_reddit_host_path(s: &str) -> String { - let (host, path) = s.split_once('/').unwrap_or((s, "")); - let host = normalize_host(host); - if path.is_empty() { - host - } else { - format!("{host}/{path}") - } -} - -/// Lowercase subreddit segment, drop listing suffixes, drop title slug after post id. -fn normalize_reddit_path(path: &str) -> String { - let mut parts: Vec = path.split('/').map(str::to_string).collect(); - if parts.len() >= 3 && parts[1] == "r" { - parts[2] = parts[2].to_ascii_lowercase(); - } - if let Some(i) = parts.iter().position(|p| p == "comments") { - if parts.len() > i + 2 { - parts.truncate(i + 2); - } - } else if parts.len() > 3 && parts.get(1).map(|s| s.as_str()) == Some("r") { - // reddit.com/r/{sub}/hot → reddit.com/r/{sub} - parts.truncate(3); - } - parts.join("/") -} - impl fmt::Display for ItemId { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { f.write_str(&self.0) @@ -268,43 +189,59 @@ mod tests { "https://old.reddit.com/r/AmItheAsshole/comments/1trnvdl/aita_for_cancelling/", ) .unwrap(); - assert_eq!(id.as_str(), "reddit.com/r/amitheasshole/comments/1trnvdl"); + assert_eq!( + id.as_str(), + "https://reddit.com/r/amitheasshole/comments/1trnvdl" + ); } #[test] fn from_url_strips_query() { let id = ItemId::from_url("https://www.reddit.com/r/rust/?sort=top").unwrap(); - assert_eq!(id.as_str(), "reddit.com/r/rust"); + assert_eq!(id.as_str(), "https://reddit.com/r/rust"); } #[test] fn from_url_short_path() { assert_eq!( ItemId::from_url("r/rust").unwrap().as_str(), - "reddit.com/r/rust" + "https://reddit.com/r/rust" ); } #[test] fn parent_of_post_is_subreddit() { - let id = ItemId::parse("reddit.com/r/aww/comments/1trnvdl").unwrap(); - assert_eq!(id.parent().unwrap().as_str(), "reddit.com/r/aww"); + let id = ItemId::from_url("https://reddit.com/r/aww/comments/1trnvdl").unwrap(); + assert_eq!(id.parent().unwrap().as_str(), "https://reddit.com/r/aww"); } #[test] fn parent_of_subreddit_is_r_segment() { - let id = ItemId::parse("reddit.com/r/movies").unwrap(); - assert_eq!(id.parent().unwrap().as_str(), "reddit.com/r"); + let id = ItemId::from_url("https://reddit.com/r/movies").unwrap(); + assert_eq!(id.parent().unwrap().as_str(), "https://reddit.com/r"); + } + + #[test] + fn breadcrumb_paths_skip_phantom_comments() { + let id = ItemId::from_url("https://reddit.com/r/aww/comments/1trnvdl").unwrap(); + let crumbs = id.breadcrumb_paths(); + let paths: Vec<_> = crumbs.iter().map(|p| p.as_str()).collect(); + assert!(!paths.iter().any(|p| p.ends_with("/comments"))); + assert!(paths.contains(&"https://reddit.com/r/aww")); } #[test] - fn breadcrumb_paths() { - let id = ItemId::parse("reddit.com/r/movies").unwrap(); + fn breadcrumb_paths_subreddit() { + let id = ItemId::from_url("https://reddit.com/r/movies").unwrap(); let crumbs = id.breadcrumb_paths(); let paths: Vec<_> = crumbs.iter().map(|p| p.as_str()).collect(); assert_eq!( paths, - vec!["reddit.com", "reddit.com/r", "reddit.com/r/movies"] + vec![ + "https://reddit.com", + "https://reddit.com/r", + "https://reddit.com/r/movies" + ] ); } @@ -312,33 +249,33 @@ mod tests { fn legacy_scope_maps_to_reddit_sub() { assert_eq!( ItemId::from_legacy_scope("rust").as_str(), - "reddit.com/r/rust" + "https://reddit.com/r/rust" ); assert!(ItemId::from_legacy_scope("").is_root()); } #[test] - fn browse_href_wraps_canonical_path() { - let id = ItemId::parse("reddit.com/r/rust").unwrap(); + fn browse_href_wraps_canonical_url() { + let id = ItemId::from_url("https://reddit.com/r/rust").unwrap(); assert_eq!(id.browse_href(), "/~/https://reddit.com/r/rust"); } #[test] fn from_browse_tail_parses_full_url() { let id = ItemId::from_browse_tail("https://reddit.com/r/AmITheAsshole"); - assert_eq!(id.as_str(), "reddit.com/r/amitheasshole"); + assert_eq!(id.as_str(), "https://reddit.com/r/amitheasshole"); } #[test] fn from_storage_strips_post_title_slug() { let id = ItemId::from_storage("reddit.com/r/rust/comments/aaa/announcing_rust_199").unwrap(); - assert_eq!(id.as_str(), "reddit.com/r/rust/comments/aaa"); + assert_eq!(id.as_str(), "https://reddit.com/r/rust/comments/aaa"); } #[test] fn from_browse_uri_strips_prefix() { let id = ItemId::from_browse_uri("/~/https://reddit.com/r/rust").unwrap(); - assert_eq!(id.as_str(), "reddit.com/r/rust"); + assert_eq!(id.as_str(), "https://reddit.com/r/rust"); } } diff --git a/server/src/projection_apply.rs b/server/src/projection_apply.rs index ebe122e417bda1d9369a53443de93d28213c5a0d..5644557a41b3e9497c7421b444155ae629fa79f1 100644 --- a/server/src/projection_apply.rs +++ b/server/src/projection_apply.rs @@ -19,13 +19,21 @@ use crate::{ storage_schema::{ensure_path_writes, entity_view_writes, vote_writes}, }; -/// Legacy-compatible scope parsing for persisted vote events. +fn parse_event_id(id: &str) -> Result { + ItemId::from_storage(id) + .or_else(|| ItemId::parse(id)) + .ok_or_else(|| EventLogError::Apply(format!("invalid id: {id}"))) +} + +/// Scope key from a vote event (canonicalized at apply time). fn parent_from_event_scope(scope: &str) -> ItemId { - if scope.contains('/') { - ItemId::parse(scope).unwrap_or_else(|| ItemId::from_legacy_scope(scope)) - } else { - ItemId::from_legacy_scope(scope) + let s = scope.trim(); + if s.is_empty() { + return ItemId::root(); } + ItemId::from_storage(s) + .or_else(|| ItemId::parse(s)) + .unwrap_or_else(|| ItemId::from_legacy_scope(s)) } pub fn apply_records( @@ -68,15 +76,11 @@ pub fn apply_records( vote_parents.insert(parent); } Event::NodeEnsured { id } => { - let parsed = ItemId::parse(id) - .or_else(|| ItemId::from_url(id)) - .ok_or_else(|| EventLogError::Apply(format!("invalid node id: {id}")))?; + let parsed = parse_event_id(id)?; ensure_path_writes(&mut batch, &parsed); } Event::EntityImported { id, payload, .. } => { - let parsed = ItemId::parse(id) - .or_else(|| ItemId::from_url(id)) - .ok_or_else(|| EventLogError::Apply(format!("invalid entity id: {id}")))?; + let parsed = parse_event_id(id)?; let view = entity_view_from_payload(&parsed, payload); entity_view_writes(&mut batch, &parsed, view.as_ref()); entity_store @@ -92,7 +96,6 @@ pub fn apply_records( .commit_with(durable::Durability::DisableWal) .map_err(|e| EventLogError::Apply(e.to_string()))?; - // Cap recent-vote windows (idempotent, blind; not part of the cursor batch). for parent in vote_parents { projection_store .trim_recent_votes(&parent) diff --git a/server/src/reddit.rs b/server/src/reddit.rs index 72caf7c33b9dd44ea15b62b59e91227cf96a3431..20b7f9e3f8be39268a1767d09f5cf81eaa6ae0df 100644 --- a/server/src/reddit.rs +++ b/server/src/reddit.rs @@ -183,7 +183,7 @@ pub fn entity_view_from_payload( id: &ItemId, payload: &Value, ) -> Option { - if id.as_str().starts_with("reddit.com") { + if id.as_str().contains("reddit.com") { return parse_reddit_view(id, payload); } None @@ -495,41 +495,59 @@ fn rate_limit_reset_secs(resp: &reqwest::Response) -> u64 { .unwrap_or(5) } +fn reddit_path_segments(id: &ItemId) -> Option> { + let s = id.as_str(); + let rest = s + .strip_prefix("https://reddit.com/") + .or_else(|| s.strip_prefix("http://reddit.com/")) + .or_else(|| s.strip_prefix("reddit.com/"))?; + let segments: Vec = rest + .split('/') + .filter(|p| !p.is_empty()) + .map(str::to_string) + .collect(); + Some(segments) +} + pub fn map_item_to_reddit_api(id: &ItemId, api_base: &str) -> String { - let path = id.as_str(); - if !path.starts_with("reddit.com/") && path != "reddit.com" { - return String::new(); - } + let segments = match reddit_path_segments(id) { + Some(s) => s, + None if matches!( + id.as_str(), + "https://reddit.com" | "http://reddit.com" | "reddit.com" + ) => + { + return String::new(); + } + None => return String::new(), + }; let base = api_base.trim_end_matches('/'); - let segments: Vec<&str> = path.split('/').collect(); - - if let Some(i) = segments.iter().position(|&p| p == "comments") { + if let Some(i) = segments.iter().position(|p| p == "comments") { if segments.len() > i + 1 { - let api_path = segments[1..=i + 1].join("/"); + let api_path = segments[..=i + 1].join("/"); return format!("{base}/{api_path}.json?raw_json=1"); } } - if segments.len() == 3 && segments[1] == "r" { - return format!("{base}/r/{}/about.json?raw_json=1", segments[2]); + if segments.len() == 2 && segments[0] == "r" { + return format!("{base}/r/{}/about.json?raw_json=1", segments[1]); } String::new() } /// Listing URL for a node's children. Currently only subreddits -/// (`reddit.com/r/` → `/r/.json`) expose a child listing. +/// (`https://reddit.com/r/` → `/r/.json`) expose a child listing. pub fn map_children_url(id: &ItemId, api_base: &str) -> String { - let path = id.as_str(); - if !path.starts_with("reddit.com/") { - return String::new(); - } + let segments = match reddit_path_segments(id) { + Some(s) => s, + None => return String::new(), + }; let base = api_base.trim_end_matches('/'); - let segments: Vec<&str> = path.split('/').collect(); - if segments.len() == 3 && segments[1] == "r" { - return format!("{base}/r/{}.json?raw_json=1&limit=25", segments[2]); + if segments.len() == 2 && segments[0] == "r" { + return format!("{base}/r/{}.json?raw_json=1&limit=25", segments[1]); } String::new() } @@ -548,8 +566,8 @@ fn parse_children(_parent: &ItemId, payload: &Value) -> Vec<(ItemId, Value)> { Some(p) if !p.is_empty() => p, _ => continue, }; - let path = format!("reddit.com{}", permalink.trim_end_matches('/')); - if let Some(id) = ItemId::from_storage(&path) { + let raw = format!("https://reddit.com{}", permalink.trim_end_matches('/')); + if let Some(id) = ItemId::from_url(&raw) { out.push((id, child.clone())); } } @@ -683,7 +701,7 @@ mod tests { #[test] fn map_subreddit_about_url() { - let id = ItemId::parse("reddit.com/r/rust").unwrap(); + let id = ItemId::from_url("https://reddit.com/r/rust").unwrap(); assert_eq!( map_item_to_reddit_api(&id, "https://www.reddit.com"), "https://www.reddit.com/r/rust/about.json?raw_json=1" @@ -699,7 +717,8 @@ mod tests { let json = include_str!("../../test/fixtures/reddit/r_rust_about.json"); let v: Value = serde_json::from_str(json).unwrap(); let entity = - entity_view_from_payload(&ItemId::parse("reddit.com/r/rust").unwrap(), &v).unwrap(); + entity_view_from_payload(&ItemId::from_url("https://reddit.com/r/rust").unwrap(), &v) + .unwrap(); assert_eq!(entity.title, "The Rust Programming Language"); } @@ -707,7 +726,8 @@ mod tests { fn parse_post_listing_extracts_thumb_and_full_preview() { let json = include_str!("../../test/fixtures/reddit/post_preview.json"); let v: Value = serde_json::from_str(json).unwrap(); - let id = ItemId::parse("reddit.com/r/nsfw/comments/1tpy6a1/angel_eyes").unwrap(); + let id = + ItemId::from_url("https://reddit.com/r/nsfw/comments/1tpy6a1/angel_eyes").unwrap(); let entity = entity_view_from_payload(&id, &v).unwrap(); assert_eq!(entity.title, "Angel Eyes"); assert!(entity.thumb_url.as_ref().unwrap().contains("width=140")); diff --git a/server/src/reducer.rs b/server/src/reducer.rs index 6578f64a41726845517cdbf59a359c69e0aa56db..5179ddeca7a7cb0fb92dfc4aa9d5a80bd9125611 100644 --- a/server/src/reducer.rs +++ b/server/src/reducer.rs @@ -248,16 +248,18 @@ mod from_recorded_tests { #[test] fn ensure_path_wires_children() { let mut tree = GlobalTree::new(); - let id = ItemId::parse("reddit.com/r/rust").unwrap(); + let id = ItemId::from_url("https://reddit.com/r/rust").unwrap(); tree.ensure_path(&id); let root = tree.get(&ItemId::root()).unwrap(); assert!(root .children - .contains(&ItemId::parse("reddit.com").unwrap())); - let reddit = tree.get(&ItemId::parse("reddit.com").unwrap()).unwrap(); + .contains(&ItemId::from_url("https://reddit.com").unwrap())); + let reddit = tree + .get(&ItemId::from_url("https://reddit.com").unwrap()) + .unwrap(); assert!(reddit .children - .contains(&ItemId::parse("reddit.com/r").unwrap())); + .contains(&ItemId::from_url("https://reddit.com/r").unwrap())); let sub = tree.get(&id).unwrap(); assert_eq!(sub.id, id); } diff --git a/server/src/render/reddit.rs b/server/src/render/reddit.rs index 7f840aa33b734a31d8cf3341a0581c8bcb9bbcf3..595e202436040b0bfc419e68f083f94757ba5d0c 100644 --- a/server/src/render/reddit.rs +++ b/server/src/render/reddit.rs @@ -9,7 +9,7 @@ use crate::{ }; pub fn is_reddit_post(id: &ItemId) -> bool { - id.as_str().starts_with("reddit.com/") && id.as_str().contains("/comments/") + id.as_str().contains("reddit.com/") && id.as_str().contains("/comments/") } /// Post detail card (inside [`crate::fetch::html::entity_panel`]). diff --git a/server/src/state.rs b/server/src/state.rs index 78126f08d90f8069a586279d79258c27a9f9f7a4..513b329ef45fc332e63b3f8ed8498a1f59feb07c 100644 --- a/server/src/state.rs +++ b/server/src/state.rs @@ -217,15 +217,21 @@ impl AppState { ratio_right: i32, ) -> Result<(), String> { let ts = crate::html::now_ms(); - let vote = VoteData::from_recorded(ts, a, b, ratio_left, ratio_right) - .ok_or_else(|| "invalid vote: need two distinct non-empty items".to_string())?; + let a_raw = a.trim(); + let b_raw = b.trim(); + if a_raw.is_empty() || b_raw.is_empty() || a_raw == b_raw { + return Err("invalid vote: need two distinct non-empty items".to_string()); + } + // Validate items canonicalize (or are opaque keys) before append. + let _ = VoteData::from_recorded(ts, a_raw, b_raw, ratio_left, ratio_right) + .ok_or_else(|| "invalid vote: need two distinct parseable items".to_string())?; let event = Event::VoteRecorded { ts, - a: vote.a.as_str().to_string(), - b: vote.b.as_str().to_string(), - ratio_left: vote.ratio_left, - ratio_right: vote.ratio_right, + a: a_raw.to_string(), + b: b_raw.to_string(), + ratio_left, + ratio_right, scope: parent.as_str().to_string(), }; @@ -253,7 +259,7 @@ mod tests { let log = EventLog::new(log_path.to_string_lossy().into_owned()); let payload = json!({"kind":"t5","data":{"title":"Rust","display_name":"rust"}}); let event = Event::EntityImported { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), ts: 1, payload: payload.clone(), }; @@ -266,14 +272,14 @@ mod tests { .await .unwrap(); let tree = projection_store - .scope_tree(&ItemId::parse("reddit.com/r/rust").unwrap()) + .scope_tree(&ItemId::parse("https://reddit.com/r/rust").unwrap()) .unwrap(); let node = tree - .get(&ItemId::parse("reddit.com/r/rust").unwrap()) + .get(&ItemId::parse("https://reddit.com/r/rust").unwrap()) .unwrap(); assert_eq!(node.data.as_ref().unwrap().title, "Rust"); let stored = entity_store - .get(&ItemId::parse("reddit.com/r/rust").unwrap()) + .get(&ItemId::parse("https://reddit.com/r/rust").unwrap()) .unwrap() .unwrap(); assert_eq!(stored["data"]["display_name"], "rust"); @@ -289,13 +295,13 @@ mod tests { event_record( 1, Event::NodeEnsured { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), }, ), event_record( 2, Event::EntityImported { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), ts: 2, payload: payload.clone(), }, @@ -325,7 +331,7 @@ mod tests { &[event_record( 1, Event::NodeEnsured { - id: "reddit.com/r/stale".into(), + id: "https://reddit.com/r/stale".into(), }, )], ) @@ -352,11 +358,11 @@ mod tests { let root = tree.get(&ItemId::root()).unwrap(); assert!(root.children.contains(&ItemId::parse("alpha").unwrap())); assert!(projection_store - .load_node(&ItemId::parse("reddit.com/r/stale").unwrap()) + .load_node(&ItemId::parse("https://reddit.com/r/stale").unwrap()) .unwrap() .is_none()); let stored = entity_store - .get(&ItemId::parse("reddit.com/r/rust").unwrap()) + .get(&ItemId::parse("https://reddit.com/r/rust").unwrap()) .unwrap() .unwrap(); assert_eq!(stored["data"]["display_name"], "rust"); @@ -370,7 +376,7 @@ mod tests { log.append(&event_record( 1, Event::NodeEnsured { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), }, )) .await @@ -387,7 +393,7 @@ mod tests { &[event_record( 2, Event::NodeEnsured { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), }, )], ) @@ -454,7 +460,7 @@ mod tests { port: 0, }) .await; - let id = ItemId::parse("reddit.com/r/rust").unwrap(); + let id = ItemId::parse("https://reddit.com/r/rust").unwrap(); state.ensure_node(&id).await.unwrap(); @@ -465,11 +471,11 @@ mod tests { let projected = state.projection_store.load_tree().unwrap(); assert!(projected.get(&id).is_some()); let reddit = projected - .get(&ItemId::parse("reddit.com").unwrap()) + .get(&ItemId::from_url("https://reddit.com").unwrap()) .unwrap(); assert!(reddit .children - .contains(&ItemId::parse("reddit.com/r").unwrap())); + .contains(&ItemId::from_url("https://reddit.com/r").unwrap())); } #[tokio::test] @@ -510,7 +516,7 @@ mod tests { log.append(&event_record( 1, Event::NodeEnsured { - id: "reddit.com/r/rust".into(), + id: "https://reddit.com/r/rust".into(), }, )) .await @@ -518,7 +524,7 @@ mod tests { log.append(&event_record( 2, Event::NodeEnsured { - id: "reddit.com/r/python".into(), + id: "https://reddit.com/r/python".into(), }, )) .await @@ -544,13 +550,13 @@ mod tests { 2 ); let tree = second - .scope_tree(&ItemId::parse("reddit.com/r/rust").unwrap()) + .scope_tree(&ItemId::parse("https://reddit.com/r/rust").unwrap()) .unwrap(); assert!(tree - .get(&ItemId::parse("reddit.com/r/rust").unwrap()) + .get(&ItemId::parse("https://reddit.com/r/rust").unwrap()) .is_some()); assert!(tree - .get(&ItemId::parse("reddit.com/r/python").unwrap()) + .get(&ItemId::parse("https://reddit.com/r/python").unwrap()) .is_none()); } @@ -611,7 +617,7 @@ mod tests { #[test] fn parse_item_param_from_url() { let id = parse_item_param("https://reddit.com/r/rust"); - assert_eq!(id.as_str(), "reddit.com/r/rust"); + assert_eq!(id.as_str(), "https://reddit.com/r/rust"); } #[test] diff --git a/server/src/url_rules/engine.rs b/server/src/url_rules/engine.rs new file mode 100644 index 0000000000000000000000000000000000000000..e29b6b48c08deb7bffe031b1e542b1e25a7bef15 --- /dev/null +++ b/server/src/url_rules/engine.rs @@ -0,0 +1,187 @@ +//! Composable URL normalization primitives. + +use std::collections::HashMap; + +use url::Url; + +/// Mutable URL view used by rule combinators before serializing to a canonical string. +#[derive(Debug, Clone)] +pub struct ParsedUrl { + pub scheme: String, + pub host: String, + pub path_segments: Vec, + pub query: HashMap, + pub fragment: Option, +} + +impl ParsedUrl { + pub fn parse(raw: &str) -> Option { + let trimmed = raw.trim(); + if trimmed.is_empty() { + return None; + } + + let with_scheme = if trimmed.contains("://") { + trimmed.to_string() + } else if trimmed.starts_with("r/") || trimmed.starts_with("/r/") { + let rest = trimmed.trim_start_matches('/').trim_start_matches("r/"); + format!("https://reddit.com/r/{rest}") + } else if trimmed.contains('.') && !trimmed.starts_with('/') { + format!("https://{trimmed}") + } else { + trimmed.to_string() + }; + + let url = Url::parse(&with_scheme).ok()?; + let host = url.host_str()?.to_string(); + let path_segments: Vec = url + .path_segments() + .map(|segs| segs.filter(|s| !s.is_empty()).map(str::to_string).collect()) + .unwrap_or_default(); + + let mut query = HashMap::new(); + for (k, v) in url.query_pairs() { + query.insert(k.into_owned(), v.into_owned()); + } + + Some(Self { + scheme: url.scheme().to_string(), + path_segments, + query, + fragment: url.fragment().map(str::to_string), + host, + }) + } + + pub fn with_path_segments(&self, segments: &[String]) -> Self { + let mut u = self.clone(); + u.path_segments = segments.to_vec(); + u + } + + pub fn to_url(&self) -> Option { + let mut url = if self.path_segments.is_empty() { + Url::parse(&format!("{}://{}", self.scheme, self.host)).ok()? + } else { + let path = format!("/{}", self.path_segments.join("/")); + Url::parse(&format!("{}://{}{}", self.scheme, self.host, path)).ok()? + }; + if !self.query.is_empty() { + let mut pairs: Vec<_> = self.query.iter().collect(); + pairs.sort_by(|a, b| a.0.cmp(b.0)); + url.query_pairs_mut().clear(); + for (k, v) in pairs { + url.query_pairs_mut().append_pair(k, v); + } + } + if let Some(ref frag) = self.fragment { + url.set_fragment(Some(frag)); + } + Some(url) + } + + pub fn canonical_string(&self) -> Option { + let url = self.to_url()?; + let mut s = url.to_string(); + if self.path_segments.is_empty() { + s = s.trim_end_matches('/').to_string(); + } + Some(s) + } +} + +pub fn force_https(u: &mut ParsedUrl) { + if u.scheme == "http" { + u.scheme = "https".to_string(); + } +} + +pub fn drop_fragment(u: &mut ParsedUrl) { + u.fragment = None; +} + +pub fn strip_www(u: &mut ParsedUrl) { + if u.host.starts_with("www.") { + u.host = u.host[4..].to_string(); + } +} + +pub fn lowercase_host(u: &mut ParsedUrl) { + u.host = u.host.to_ascii_lowercase(); +} + +pub fn lowercase_path(u: &mut ParsedUrl) { + for seg in &mut u.path_segments { + *seg = seg.to_ascii_lowercase(); + } +} + +pub fn clear_query(u: &mut ParsedUrl) { + u.query.clear(); +} + +pub fn keep_only_query(u: &mut ParsedUrl, keys: &[&str]) { + u.query + .retain(|k, _| keys.iter().any(|want| want == &k.as_str())); +} + +pub fn strip_tracking_params(u: &mut ParsedUrl) { + u.query.retain(|k, _| { + let lower = k.to_ascii_lowercase(); + !(lower.starts_with("utm_") + || matches!( + lower.as_str(), + "fbclid" | "gclid" | "ref" | "ref_src" | "ref_source" | "mc_cid" | "mc_eid" + )) + }); +} + +pub fn truncate_after_segment(u: &mut ParsedUrl, name: &str, keep: usize) { + if let Some(i) = u.path_segments.iter().position(|s| s == name) { + let end = (i + 1 + keep).min(u.path_segments.len()); + u.path_segments.truncate(end); + } +} + +pub fn drop_listing_suffix(u: &mut ParsedUrl, suffixes: &[&str]) { + if u.path_segments.len() >= 3 && u.path_segments.first().map(String::as_str) == Some("r") { + if let Some(last) = u.path_segments.last() { + if suffixes.iter().any(|s| *s == last.as_str()) { + u.path_segments.pop(); + } + } + } +} + +pub fn normalize_reddit_host(u: &mut ParsedUrl) { + if matches!( + u.host.as_str(), + "old.reddit.com" | "new.reddit.com" | "www.reddit.com" + ) { + u.host = "reddit.com".to_string(); + } +} + +pub fn rewrite_youtu_be(u: &mut ParsedUrl) { + if u.host == "youtu.be" && u.path_segments.len() == 1 { + let id = u.path_segments[0].clone(); + u.host = "youtube.com".to_string(); + u.path_segments = vec!["watch".to_string()]; + u.query.insert("v".to_string(), id); + } +} + +pub fn rewrite_youtube_shorts(u: &mut ParsedUrl) { + if u.host == "youtube.com" && u.path_segments.first().map(String::as_str) == Some("shorts") { + if let Some(id) = u.path_segments.get(1).cloned() { + u.path_segments = vec!["watch".to_string()]; + u.query.insert("v".to_string(), id); + } + } +} + +pub fn normalize_youtube_host(u: &mut ParsedUrl) { + if matches!(u.host.as_str(), "m.youtube.com" | "www.youtube.com") { + u.host = "youtube.com".to_string(); + } +} diff --git a/server/src/url_rules/mod.rs b/server/src/url_rules/mod.rs new file mode 100644 index 0000000000000000000000000000000000000000..03d53bd3e82d704a01ba3fd8dd02b7d31422c0de --- /dev/null +++ b/server/src/url_rules/mod.rs @@ -0,0 +1,13 @@ +//! URL canonicalization and hierarchy rules for [`crate::path_types::ItemId`]. + +mod engine; +mod registry; + +pub use registry::{ + canonicalize_raw, looks_like_url, navigable_breadcrumbs, parent_url, resolve_id, CanonicalResult, +}; + +/// Resolve raw input to canonical URL. +pub fn resolve_canonical(raw: &str) -> Option { + canonicalize_raw(raw.trim()).map(|r| r.canonical) +} diff --git a/server/src/url_rules/registry.rs b/server/src/url_rules/registry.rs new file mode 100644 index 0000000000000000000000000000000000000000..14514e9af8385fb2b9b2f35eb9ee14d453d4b97c --- /dev/null +++ b/server/src/url_rules/registry.rs @@ -0,0 +1,235 @@ +//! Per-domain canonicalization and hierarchy rules. + +use std::collections::HashSet; + +use super::engine::{ + clear_query, drop_fragment, drop_listing_suffix, force_https, keep_only_query, lowercase_host, + lowercase_path, normalize_reddit_host, normalize_youtube_host, rewrite_youtu_be, + rewrite_youtube_shorts, strip_tracking_params, strip_www, truncate_after_segment, ParsedUrl, +}; + +/// Result of canonicalizing a raw URL string. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CanonicalResult { + pub canonical: String, + /// When the input normalizes to a different string, the original is an alias. + pub alias_of: Option, +} + +fn apply_global(u: &mut ParsedUrl) { + force_https(u); + drop_fragment(u); + strip_www(u); + lowercase_host(u); + strip_tracking_params(u); +} + +fn normalize_reddit(u: &mut ParsedUrl) { + normalize_reddit_host(u); + lowercase_path(u); + truncate_after_segment(u, "comments", 1); + drop_listing_suffix(u, &["hot", "top", "new", "rising", "controversial"]); + clear_query(u); +} + +fn normalize_youtube(u: &mut ParsedUrl) { + rewrite_youtu_be(u); + normalize_youtube_host(u); + rewrite_youtube_shorts(u); + keep_only_query(u, &["v", "list"]); +} + +fn normalize_default(_u: &mut ParsedUrl) { + // Global rules only. +} + +fn domain_key(host: &str) -> &'static str { + if host == "reddit.com" || host.ends_with(".reddit.com") { + "reddit.com" + } else if host == "youtube.com" || host == "youtu.be" { + "youtube.com" + } else { + "default" + } +} + +fn normalize_for_host(u: &mut ParsedUrl) { + apply_global(u); + match domain_key(&u.host) { + "reddit.com" => normalize_reddit(u), + "youtube.com" => normalize_youtube(u), + _ => normalize_default(u), + } +} + +/// Structural path segments that must not become standalone tree nodes when more path follows. +fn structural_trailing(host: &str) -> &'static [&'static str] { + match domain_key(host) { + "reddit.com" => &["comments"], + _ => &[], + } +} + +/// Canonicalize a raw URL. Returns `None` if the input is not URL-like. +pub fn canonicalize_raw(raw: &str) -> Option { + let trimmed = raw.trim(); + if trimmed.is_empty() { + return None; + } + let mut u = ParsedUrl::parse(trimmed)?; + let input_snapshot = u.canonical_string()?; + normalize_for_host(&mut u); + let canonical = u.canonical_string()?; + let alias_of = if input_snapshot != canonical { + Some(trimmed.to_string()) + } else { + None + }; + Some(CanonicalResult { + canonical, + alias_of, + }) +} + +/// Resolve a stored or event id string to its canonical URL identity. +pub fn resolve_id(raw: &str) -> Option { + canonicalize_raw(raw).map(|r| r.canonical) +} + +/// Navigable ancestor URLs from domain root up to and including `canonical` (full URLs). +pub fn navigable_breadcrumbs(canonical: &str) -> Vec { + let Some(u) = ParsedUrl::parse(canonical) else { + return vec![canonical.to_string()]; + }; + let structural: HashSet<&str> = structural_trailing(&u.host).iter().copied().collect(); + let n = u.path_segments.len(); + let mut out = Vec::new(); + + // Domain root (no path segments). + if let Some(base) = u.with_path_segments(&[]).canonical_string() { + out.push(base); + } + + for i in 0..n { + let segs: Vec = u.path_segments[..=i].to_vec(); + let is_last = i == n - 1; + let seg = u.path_segments[i].as_str(); + if structural.contains(seg) && !is_last { + continue; + } + if let Some(url) = u.with_path_segments(&segs).canonical_string() { + if out.last() != Some(&url) { + out.push(url); + } + } + } + out +} + +/// Immediate parent scope URL, or `None` for tree root / opaque single-segment ids. +pub fn parent_url(canonical: &str) -> Option { + let crumbs = navigable_breadcrumbs(canonical); + if crumbs.len() <= 1 { + None + } else { + crumbs.get(crumbs.len() - 2).cloned() + } +} + +/// True when `raw` looks like a URL (has scheme or host-like shape). +pub fn looks_like_url(raw: &str) -> bool { + let t = raw.trim(); + t.contains("://") + || t.starts_with("r/") + || t.starts_with("/r/") + || (t.contains('.') && t.contains('/')) + || t.starts_with("reddit.com") + || t.starts_with("www.") + || t.starts_with("youtu.be/") +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn reddit_post_drops_slug_and_normalizes_host() { + let r = canonicalize_raw( + "https://old.reddit.com/r/AmItheAsshole/comments/1trnvdl/aita_for_cancelling/", + ) + .unwrap(); + assert_eq!( + r.canonical, + "https://reddit.com/r/amitheasshole/comments/1trnvdl" + ); + } + + #[test] + fn reddit_strips_query_and_listing() { + assert_eq!( + canonicalize_raw("https://www.reddit.com/r/rust/?sort=top") + .unwrap() + .canonical, + "https://reddit.com/r/rust" + ); + assert_eq!( + canonicalize_raw("https://www.reddit.com/r/programming/hot") + .unwrap() + .canonical, + "https://reddit.com/r/programming" + ); + } + + #[test] + fn reddit_short_path() { + assert_eq!( + canonicalize_raw("r/rust").unwrap().canonical, + "https://reddit.com/r/rust" + ); + } + + #[test] + fn reddit_breadcrumbs_skip_phantom_comments() { + let post = "https://reddit.com/r/aww/comments/1trnvdl"; + let crumbs = navigable_breadcrumbs(post); + assert!(!crumbs.iter().any(|c| c.ends_with("/comments"))); + assert_eq!( + crumbs.last().map(String::as_str), + Some(post) + ); + assert!(crumbs.contains(&"https://reddit.com/r/aww".to_string())); + } + + #[test] + fn reddit_parent_of_post_is_subreddit() { + assert_eq!( + parent_url("https://reddit.com/r/aww/comments/1trnvdl").as_deref(), + Some("https://reddit.com/r/aww") + ); + } + + #[test] + fn youtube_youtu_be_and_watch_same_canonical() { + let a = canonicalize_raw("https://youtu.be/dQw4w9WgXcQ").unwrap().canonical; + let b = canonicalize_raw("https://www.youtube.com/watch?v=dQw4w9WgXcQ&t=10").unwrap(); + assert_eq!(a, b.canonical); + assert_eq!(a, "https://youtube.com/watch?v=dQw4w9WgXcQ"); + } + + #[test] + fn legacy_schemeless_upgrades() { + assert_eq!( + canonicalize_raw("reddit.com/r/rust/comments/aaa/announcing_rust_199") + .unwrap() + .canonical, + "https://reddit.com/r/rust/comments/aaa" + ); + } + + #[test] + fn alias_recorded_when_input_differs() { + let r = canonicalize_raw("https://youtu.be/abc123").unwrap(); + assert_eq!(r.canonical, "https://youtube.com/watch?v=abc123"); + assert!(r.alias_of.is_some()); + } +}