You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [af73743d] Replace GroupState with ScopeVotes and derive edges at ranking time. Store only uuid_votes and recent_votes per scope; rank centrality and pair logic rebuild edge weights on demand instead of maintaining cached state. Co-authored-by: Cursor Side A — unified diff (full patch): diff --git a/server/src/events.rs b/server/src/events.rs index 8a166d49b4f26835fbc2b58cb1f4bdbf002763b8..015208311f6c5c23a0e8aab068d002a69c89c4e1 100644 --- a/server/src/events.rs +++ b/server/src/events.rs @@ -43,7 +43,7 @@ pub enum ViewEvent { #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(tag = "type", rename_all = "snake_case")] pub enum Event { - /// Pairwise comparison vote (replayed into the parent node's [`crate::reducer::GroupState`] on boot). + /// Pairwise comparison vote (replayed into the parent node's [`crate::reducer::ScopeVotes`] on boot). /// `scope` is the parent [`crate::path_types::ItemId`] string; empty string is the tree root. VoteRecorded { ts: i64, diff --git a/server/src/html/mod.rs b/server/src/html/mod.rs index 4eff2e19ed4d303ff8e80c1eabd8a15b4990e643..1e2e7a06856d8a62378741aaf5ed94a4ffed337e 100644 --- a/server/src/html/mod.rs +++ b/server/src/html/mod.rs @@ -13,7 +13,7 @@ use crate::{ form_template::template_json_compact, path_types::ItemId, ranking::{ - connected_components_from_voted_pairs, ranked_items_subset, RankedItem, MAX_ITERS, TOL, + ranked_items_subset, scope_components, RankedItem, MAX_ITERS, TOL, }, reducer::{GlobalTree, NodeState}, state::AppState, @@ -397,10 +397,9 @@ pub fn ranking_panel_with_highlights( tree: &GlobalTree, highlighted: &HashSet, ) -> Markup { - let group = &node.local_ranking; - let n = group.idx_to_item.len(); - let (comps, _isolates) = - connected_components_from_voted_pairs(n, group.voted_pairs.iter().copied()); + let scope = &node.votes; + let (comps, _isolates, _) = + scope_components(scope); // Each connected component of voted items is its own ranking; isolated and // never-voted children fall into the "unranked" bucket below. @@ -410,7 +409,7 @@ pub fn ranking_panel_with_highlights( if comp.len() < 2 { continue; } - let ranked = ranked_items_subset(group, comp, MAX_ITERS, TOL); + let ranked = ranked_items_subset(scope, comp, MAX_ITERS, TOL); for r in &ranked { ranked_ids.insert(r.item.clone()); } diff --git a/server/src/html/vote.rs b/server/src/html/vote.rs index bf82aef3ad9c5f4e9e877c47dab27beb29a80b8f..3aa00c417c89a9cab3417c650b50ed7c73f08e20 100644 --- a/server/src/html/vote.rs +++ b/server/src/html/vote.rs @@ -14,7 +14,7 @@ use crate::{ html::{ranking_panel_with_highlights, scope_theme_style, JsBuilder}, pair::{children_of, resolve_pair, suggest_next_pair_in_pool}, path_types::ItemId, - reducer::{GlobalTree, GroupState, NodeState, VoteData}, + reducer::{GlobalTree, NodeState, ScopeVotes, VoteData}, state::{parse_item_param, AppState}, ui_action::UI_RPC_FIELD, }; @@ -68,8 +68,8 @@ fn ratios_for_page(v: &VoteData, page_left: &ItemId, page_right: &ItemId) -> (i3 } } -fn edge_votes(group: &GroupState, left: &ItemId, right: &ItemId) -> Vec { - group +fn edge_votes(scope: &ScopeVotes, left: &ItemId, right: &ItemId) -> Vec { + scope .recent_votes .iter() .filter(|v| { @@ -113,11 +113,11 @@ fn slider_value_from_ratios(r_left: i32, r_right: i32) -> i32 { fn vote_edge_history( tree: &GlobalTree, - group: &GroupState, + scope: &ScopeVotes, left: &ItemId, right: &ItemId, ) -> Markup { - let mut votes = edge_votes(group, left, right); + let mut votes = edge_votes(scope, left, right); votes.sort_by(|a, b| b.ts.cmp(&a.ts)); let legend_left = child_title(tree, left); let legend_right = child_title(tree, right); @@ -228,9 +228,9 @@ pub(crate) fn vote_recorded_morph( ) -> JsBuilder { let pool = children_of(tree, parent); let empty = NodeState::default(); - let group = tree.get(parent).unwrap_or(&empty).local_ranking.clone(); - let edge_history = vote_edge_history(tree, &group, left, right); - let next_pair = suggest_next(&group, left, right, &pool); + let scope = tree.get(parent).unwrap_or(&empty).votes.clone(); + let edge_history = vote_edge_history(tree, &scope, left, right); + let next_pair = suggest_next(&scope, left, right, &pool); let actions = vote_compare_actions(parent, next_pair.as_ref()); let sidebar = vote_ranking_sidebar(tree, parent, left, right); JsBuilder::new() @@ -252,12 +252,12 @@ fn vote_compare_item_card(tree: &GlobalTree, item: &ItemId, side_class: &str) -> } fn suggest_next( - group: &GroupState, + scope: &ScopeVotes, left: &ItemId, right: &ItemId, pool: &[ItemId], ) -> Option<(ItemId, ItemId)> { - suggest_next_pair_in_pool(group, pool, Some((left, right))) + suggest_next_pair_in_pool(scope, pool, Some((left, right))) } pub async fn vote_page( @@ -284,9 +284,9 @@ pub async fn vote_page( }; let pool = children_of(&tree, &parent); - let group = &parent_node.local_ranking; - let next_pair = suggest_next(group, &left, &right, &pool); - let edge_history = vote_edge_history(&tree, group, &left, &right); + let scope = &parent_node.votes; + let next_pair = suggest_next(&scope, &left, &right, &pool); + let edge_history = vote_edge_history(&tree, &scope, &left, &right); let rpc_json = template_json_compact(&serde_json::json!({ "action": "record_vote", @@ -388,8 +388,8 @@ mod polarity_tests { let mut tree = GlobalTree::new(); tree.apply_vote(&parent, vote, TEST_ACTOR_UUID); - let group = &tree.get(&parent).unwrap().local_ranking; - let ranked = ranked_items(group); + let scope = &tree.get(&parent).unwrap().votes; + let ranked = ranked_items(scope); assert_eq!( ranked[0].item, left, "left item should rank first when ratio favours the left" diff --git a/server/src/pair.rs b/server/src/pair.rs index 42a1b1eb2adf16730d34d0fe23c13d5a75d7ba27..9873295c51526726089875bbfd2f97d7faa91872 100644 --- a/server/src/pair.rs +++ b/server/src/pair.rs @@ -11,8 +11,8 @@ use std::collections::{HashMap, HashSet}; use crate::{ path_types::ItemId, - ranking::{connected_components_from_voted_pairs, ranked_items}, - reducer::{GlobalTree, GroupState}, + ranking::{pair_is_voted, ranked_items, scope_components}, + reducer::{GlobalTree, ScopeVotes}, }; fn pairs_match(a: &ItemId, b: &ItemId, x: &ItemId, y: &ItemId) -> bool { @@ -23,26 +23,15 @@ fn pair_excluded(a: &ItemId, b: &ItemId, exclude: Option<(&ItemId, &ItemId)>) -> exclude.is_some_and(|(x, y)| pairs_match(a, b, x, y)) } -fn pair_is_voted(group: &GroupState, a: &ItemId, b: &ItemId) -> bool { - let Some(&ai) = group.item_to_idx.get(a) else { - return false; - }; - let Some(&bi) = group.item_to_idx.get(b) else { - return false; - }; - let (i, j) = if ai < bi { (ai, bi) } else { (bi, ai) }; - group.voted_pairs.contains(&(i, j)) -} struct ComponentLayout { ids: HashMap, established: HashSet, } -fn component_layout(group: &GroupState, pool: &[ItemId]) -> ComponentLayout { - let n = group.idx_to_item.len(); - let (comps, isolates) = - connected_components_from_voted_pairs(n, group.voted_pairs.iter().copied()); +fn component_layout(scope: &ScopeVotes, pool: &[ItemId]) -> ComponentLayout { + let (comps, isolates, idx_to_item) = scope_components(scope); + let n = idx_to_item.len(); let mut established = HashSet::new(); let mut ids: HashMap = HashMap::new(); @@ -52,14 +41,14 @@ fn component_layout(group: &GroupState, pool: &[ItemId]) -> ComponentLayout { } for &idx in comp { if idx < n { - ids.insert(group.idx_to_item[idx].clone(), comp_idx); + ids.insert(idx_to_item[idx].clone(), comp_idx); } } } let mut next = comps.len(); for &idx in &isolates { if idx < n { - ids.insert(group.idx_to_item[idx].clone(), next); + ids.insert(idx_to_item[idx].clone(), next); next += 1; } } @@ -121,7 +110,7 @@ fn established_groups_in_pool<'a>( groups } -fn ranked_pool_order(group: &GroupState, pool: &[ItemId]) -> Vec { +fn ranked_pool_order(group: &ScopeVotes, pool: &[ItemId]) -> Vec { let pool_set: HashSet<_> = pool.iter().collect(); ranked_items(group) .into_iter() @@ -132,7 +121,7 @@ fn ranked_pool_order(group: &GroupState, pool: &[ItemId]) -> Vec { /// Walk 1↔2, 2↔3, …; optional `require_unvoted` skips voted edges. fn zip_adjacent_pair( - group: &GroupState, + group: &ScopeVotes, order: &[ItemId], exclude: Option<(&ItemId, &ItemId)>, require_unvoted: bool, @@ -153,7 +142,7 @@ fn zip_adjacent_pair( /// Grow the voted graph toward one component (no rank centrality). fn suggest_grow_pair( - group: &GroupState, + group: &ScopeVotes, pool: &[ItemId], layout: &ComponentLayout, exclude: Option<(&ItemId, &ItemId)>, @@ -216,7 +205,7 @@ fn suggest_grow_pair( /// Pick the next pair to vote on within `pool`. pub fn suggest_next_pair_in_pool( - group: &GroupState, + group: &ScopeVotes, pool: &[ItemId], exclude: Option<(&ItemId, &ItemId)>, ) -> Option<(ItemId, ItemId)> { @@ -315,7 +304,7 @@ pub fn resolve_pair( (None, None) => { let group = tree .get(parent) - .map(|n| &n.local_ranking) + .map(|n| &n.votes) .cloned() .unwrap_or_default(); suggest_next_pair_in_pool(&group, &children, None).ok_or(PairError::NoPair) @@ -398,7 +387,7 @@ mod tests { "https://reddit.com/r/rust/b", ], ); - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); assert!(!pair_is_voted(&group, &pool[0], &pool[1])); assert!(suggest_next_pair_in_pool(&group, &pool, None).is_some()); @@ -417,7 +406,7 @@ mod tests { ); let vote = test_vote(1, "https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b", 2, 1); apply(&mut tree, &parent, vote); - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let (l, r) = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let voted_ab = (l.as_str() == "https://reddit.com/r/rust/a" && r.as_str() == "https://reddit.com/r/rust/b") @@ -441,7 +430,7 @@ mod tests { let cd = test_vote(2, "https://reddit.com/r/rust/c", "https://reddit.com/r/rust/d", 2, 1); apply(&mut tree, &parent, ab); apply(&mut tree, &parent, cd); - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); @@ -467,7 +456,7 @@ mod tests { ); let ab = test_vote(1, "https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b", 2, 1); apply(&mut tree, &parent, ab); - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); @@ -496,7 +485,7 @@ mod tests { ); let ab = test_vote(1, "https://reddit.com/r/rust/a", "https://reddit.com/r/rust/b", 2, 1); apply(&mut tree, &parent, ab); - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); @@ -522,7 +511,7 @@ mod tests { let v = test_vote(1, a, b, l, r); apply(&mut tree, &parent, v); } - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); @@ -550,7 +539,7 @@ mod tests { let v = test_vote(1, a, b, l, r); apply(&mut tree, &parent, v); } - let group = tree.get(&parent).unwrap().local_ranking.clone(); + let group = tree.get(&parent).unwrap().votes.clone(); let pool = children_of(&tree, &parent); let pair = suggest_next_pair_in_pool(&group, &pool, None).unwrap(); let chosen = pair_set(&pair); diff --git a/server/src/projection_store.rs b/server/src/projection_store.rs index 27fc0bb0519d9f035dceec1fc8b0fa74b00b6b40..8c1a466183a96173fe52b144fb67c2454b715fbd 100644 --- a/server/src/projection_store.rs +++ b/server/src/projection_store.rs @@ -214,7 +214,7 @@ mod tests { let loaded = store.load_tree().unwrap(); let root = loaded.get(&ItemId::root()).unwrap(); - assert_eq!(root.local_ranking.idx_to_item.len(), 2); + assert_eq!(crate::ranking::ranked_items(&root.votes).len(), 2); assert!(root.children.contains(&ItemId::opaque("alpha"))); } diff --git a/server/src/ranking.rs b/server/src/ranking.rs index 1fc3298d2b2864e9711cd262a034677e45c7090c..cc4de6da11b02135e5e6b1d2a68646cb77870b9d 100644 --- a/server/src/ranking.rs +++ b/server/src/ranking.rs @@ -1,7 +1,7 @@ -use std::collections::{HashMap, HashSet}; +use std::collections::{BTreeSet, HashMap, HashSet}; use crate::path_types::ItemId; -use crate::reducer::GroupState; +use crate::reducer::{canonical_pair_ids, ScopeVotes}; #[derive(Debug, Clone)] pub struct RankedItem { @@ -9,15 +9,76 @@ pub struct RankedItem { pub score: f64, } -/// Power-iteration cap and convergence tolerance for rank centrality. pub const MAX_ITERS: usize = 10_000; pub const TOL: f64 = 1e-8; -/// Compute connected components over the voted-pairs graph (treated as undirected). -/// -/// Returns: -/// - `components`: each component is a sorted list of node indices, excluding isolates. -/// - `isolates`: sorted list of node indices with degree 0 (no voted pairs). +pub fn item_index(scope: &ScopeVotes) -> (HashMap, Vec) { + let mut item_strs: BTreeSet = BTreeSet::new(); + for vote in scope.uuid_votes.values() { + item_strs.insert(vote.a.as_str().to_string()); + item_strs.insert(vote.b.as_str().to_string()); + } + let mut idx_to_item: Vec = Vec::with_capacity(item_strs.len()); + let mut item_to_idx: HashMap = HashMap::with_capacity(item_strs.len()); + for s in item_strs { + let id = ItemId::from_storage(&s).unwrap_or_else(|| ItemId::opaque(&s)); + let idx = idx_to_item.len(); + item_to_idx.insert(id.clone(), idx); + idx_to_item.push(id); + } + (item_to_idx, idx_to_item) +} + +pub fn edges_from_scope(scope: &ScopeVotes) -> HashMap<(usize, usize), f64> { + let (item_to_idx, _) = item_index(scope); + let mut edges: HashMap<(usize, usize), f64> = HashMap::new(); + for vote in scope.uuid_votes.values() { + let Some(&ai) = item_to_idx.get(&vote.a) else { + continue; + }; + let Some(&bi) = item_to_idx.get(&vote.b) else { + continue; + }; + let w_a = vote.ratio_left as f64 * vote.trust_weight; + let w_b = vote.ratio_right as f64 * vote.trust_weight; + if w_a > 0.0 { + *edges.entry((bi, ai)).or_insert(0.0) += w_a; + } + if w_b > 0.0 { + *edges.entry((ai, bi)).or_insert(0.0) += w_b; + } + } + edges +} + +pub fn edge_weight_sum(scope: &ScopeVotes) -> f64 { + edges_from_scope(scope).values().sum() +} + +pub fn voted_pair_indices(scope: &ScopeVotes) -> HashSet<(usize, usize)> { + let (item_to_idx, _) = item_index(scope); + let mut pairs = HashSet::new(); + for vote in scope.uuid_votes.values() { + let Some(&ai) = item_to_idx.get(&vote.a) else { + continue; + }; + let Some(&bi) = item_to_idx.get(&vote.b) else { + continue; + }; + let (i, j) = if ai < bi { (ai, bi) } else { (bi, ai) }; + pairs.insert((i, j)); + } + pairs +} + +pub fn pair_is_voted(scope: &ScopeVotes, a: &ItemId, b: &ItemId) -> bool { + let (lo, hi) = canonical_pair_ids(a, b); + scope + .uuid_votes + .keys() + .any(|(_, l, h)| l == &lo && h == &hi) +} + pub fn connected_components_from_voted_pairs( n: usize, voted_pairs: impl Iterator, @@ -63,20 +124,25 @@ pub fn connected_components_from_voted_pairs( (comps, isolates) } -/// Compute rank-centrality scores for the whole group and return items sorted -/// by score (descending). Recomputed fresh from the edge set on every call — -/// there is no score cache. -pub fn ranked_items(group: &GroupState) -> Vec { - let n = group.idx_to_item.len(); - let scores = - compute_scores_from_edges(n, group.edges.iter().map(|(&k, &w)| (k, w)), MAX_ITERS, TOL); +pub fn scope_components(scope: &ScopeVotes) -> (Vec>, Vec, Vec) { + let (_, idx_to_item) = item_index(scope); + let n = idx_to_item.len(); + let pairs = voted_pair_indices(scope); + let (comps, isolates) = connected_components_from_voted_pairs(n, pairs.into_iter()); + (comps, isolates, idx_to_item) +} - let mut items: Vec = group - .idx_to_item - .iter() +pub fn ranked_items(scope: &ScopeVotes) -> Vec { + let (_, idx_to_item) = item_index(scope); + let n = idx_to_item.len(); + let edges = edges_from_scope(scope); + let scores = compute_scores_from_edges(n, edges.into_iter(), MAX_ITERS, TOL); + + let mut items: Vec = idx_to_item + .into_iter() .enumerate() .map(|(i, item)| RankedItem { - item: item.clone(), + item, score: *scores.get(i).unwrap_or(&0.0), }) .collect(); @@ -89,11 +155,8 @@ pub fn ranked_items(group: &GroupState) -> Vec { items } -/// Highest- and lowest-ranked items for a group. Returns up to `k` items from -/// each end with no overlap. If the group has `2*k` items or fewer, `top` holds -/// the full ranking and `bottom` is empty (so nothing is shown twice). -pub fn top_bottom(group: &GroupState, k: usize) -> (Vec, Vec) { - let items = ranked_items(group); +pub fn top_bottom(scope: &ScopeVotes, k: usize) -> (Vec, Vec) { + let items = ranked_items(scope); if k == 0 || items.len() <= 2 * k { return (items, Vec::new()); } @@ -115,7 +178,6 @@ pub fn compute_scores_from_edges( return vec![1.0]; } - // Collect raw edges into a map for pairwise normalization. let mut raw: HashMap<(usize, usize), f64> = HashMap::new(); for ((src, dst), w) in edges { if src >= n || dst >= n || w <= 0.0 { @@ -124,9 +186,6 @@ pub fn compute_scores_from_edges( *raw.entry((src, dst)).or_insert(0.0) += w; } - // Pairwise normalization: a_ij = A_ij / (A_ij + A_ji). - // This ensures repeated votes on the same pair don't inflate influence - // beyond what the ratio implies. let keys: Vec<(usize, usize)> = raw.keys().copied().collect(); let mut normalized: HashMap<(usize, usize), f64> = HashMap::new(); for (i, j) in keys { @@ -145,17 +204,6 @@ pub fn compute_scores_from_edges( } } - // Rank Centrality (Negahban, Oh, Shah 2012, §3.1): - // P_ij = (1/d_max) * A_ij for i ≠ j compared - // P_ii = 1 - (1/d_max) * Σ_k A_ik - // where d_i is the *degree* (number of distinct neighbors compared) and - // d_max = max_i d_i. Using the unweighted degree — not the sum of - // pairwise-normalized weights — is what guarantees aperiodicity: it - // forces P_ii > 0 for every non-maximum-degree node, and for max-degree - // nodes whenever any neighbor weight is below 1 (i.e. not a unanimous - // loss). Without this, regular comparison graphs (e.g. a pure star at - // ratio 2:1) produce a bipartite chain that oscillates instead of - // converging — see issue #146. let mut out_edges: Vec> = vec![Vec::new(); n]; let mut neighbors: Vec> = vec![HashSet::new(); n]; @@ -213,11 +261,8 @@ pub fn compute_scores_from_edges( scores } -/// Rank-centrality within a subset of items (an induced subgraph), using the group's aggregated edges. -/// -/// `idxs` are indices into `group.idx_to_item`. The returned items use the original item names. pub fn ranked_items_subset( - group: &GroupState, + scope: &ScopeVotes, idxs: &[usize], max_iters: usize, tol: f64, @@ -226,13 +271,15 @@ pub fn ranked_items_subset( return vec![]; } - // Map original idx -> compact idx [0..m) + let (_, idx_to_item) = item_index(scope); + let edges = edges_from_scope(scope); + let mut map: HashMap = HashMap::with_capacity(idxs.len()); for (j, &i) in idxs.iter().enumerate() { map.insert(i, j); } - let edges_iter = group.edges.iter().filter_map(|(&(src, dst), &w)| { + let edges_iter = edges.into_iter().filter_map(|((src, dst), w)| { let s = *map.get(&src)?; let d = *map.get(&dst)?; Some(((s, d), w)) @@ -240,12 +287,11 @@ pub fn ranked_items_subset( let scores = compute_scores_from_edges(idxs.len(), edges_iter, max_iters, tol); - // Filter out entries where idx_to_item doesn't have the slot (shouldn't happen, but be safe). let mut items: Vec = idxs .iter() .enumerate() .filter_map(|(j, &orig)| { - let item = group.idx_to_item.get(orig)?.clone(); + let item = idx_to_item.get(orig)?.clone(); Some(RankedItem { item, score: *scores.get(j).unwrap_or(&0.0), @@ -261,8 +307,8 @@ pub fn ranked_items_subset( items } -pub fn group_summary_scores(group: &GroupState) -> HashMap { - ranked_items(group) +pub fn group_summary_scores(scope: &ScopeVotes) -> HashMap { + ranked_items(scope) .into_iter() .map(|r| (r.item, r.score)) .collect() @@ -274,32 +320,26 @@ mod tests { use crate::identity::{DEFAULT_PSEUDONYM, TEST_ACTOR_UUID}; use crate::reducer::VoteData; - fn mk_group() -> GroupState { - GroupState::new() + fn mk_scope() -> ScopeVotes { + ScopeVotes::default() } fn vote(ts: i64, a: &str, b: &str, l: i32, r: i32) -> VoteData { VoteData::from_event(ts, a, b, l, r, DEFAULT_PSEUDONYM.to_string(), 1.0).unwrap() } - fn apply(g: &mut GroupState, v: VoteData) { - g.apply_vote(v, TEST_ACTOR_UUID); + fn apply(scope: &mut ScopeVotes, v: VoteData) { + scope.apply_vote(v, TEST_ACTOR_UUID); } - /// Regression for issue #146: pure forward star at default `>` ratio (2:1). - /// Under the old (sum-of-weights) divisor every node had P_ii = 0 and the - /// chain was bipartite; power iteration oscillated and returned the - /// uniform initial distribution after an even number of steps. Using the - /// paper's degree-based d_max gives every node a positive self-loop and - /// the chain converges to the correct stationary distribution. #[test] fn star_topology_winner_at_top_via_subset() { - let mut g = mk_group(); - g.apply_vote(vote(1, "zebra", "alpha", 2, 1), TEST_ACTOR_UUID); - g.apply_vote(vote(2, "zebra", "beta", 2, 1), TEST_ACTOR_UUID); + let mut scope = mk_scope(); + apply(&mut scope, vote(1, "zebra", "alpha", 2, 1)); + apply(&mut scope, vote(2, "zebra", "beta", 2, 1)); - let mut items: Vec<(usize, String)> = g - .idx_to_item + let (_, idx_to_item) = item_index(&scope); + let mut items: Vec<(usize, String)> = idx_to_item .iter() .enumerate() .map(|(i, it)| (i, it.as_str().to_string())) @@ -307,78 +347,51 @@ mod tests { items.sort_by(|a, b| a.1.cmp(&b.1)); let idxs: Vec = items.iter().map(|(i, _)| *i).collect(); - let ranked = ranked_items_subset(&g, &idxs, 10000, 1e-8); - for r in &ranked { - eprintln!("{}: {}", r.item.as_str(), r.score); - } - assert_eq!( - ranked[0].item.as_str(), - "zebra", - "zebra won both votes and should rank #1" - ); + let ranked = ranked_items_subset(&scope, &idxs, 10000, 1e-8); + assert_eq!(ranked[0].item.as_str(), "zebra"); } #[test] fn top_bottom_splits_ends_without_overlap() { - let mut g = mk_group(); - // Chain a > b > c > d > e > f so ranks are well separated. + let mut scope = mk_scope(); for (hi, lo) in [("a", "b"), ("b", "c"), ("c", "d"), ("d", "e"), ("e", "f")] { - apply(&mut g, vote(1, hi, lo, 2, 1)); + apply(&mut scope, vote(1, hi, lo, 2, 1)); } - let (top, bottom) = top_bottom(&g, 2); + let (top, bottom) = top_bottom(&scope, 2); assert_eq!(top.len(), 2); assert_eq!(bottom.len(), 2); - // No overlap between the two ends. for t in &top { assert!(bottom.iter().all(|b| b.item != t.item)); } - // Best item ranks above the worst item. assert!(top[0].score >= bottom[bottom.len() - 1].score); } #[test] fn top_bottom_small_group_has_empty_bottom() { - let mut g = mk_group(); - apply(&mut g, vote(1, "a", "b", 2, 1)); - let (top, bottom) = top_bottom(&g, 5); + let mut scope = mk_scope(); + apply(&mut scope, vote(1, "a", "b", 2, 1)); + let (top, bottom) = top_bottom(&scope, 5); assert_eq!(top.len(), 2); assert!(bottom.is_empty()); } #[test] fn connected_components_split_disconnected_pairs() { - let mut g = mk_group(); - // Two disconnected edges: (a,b) and (c,d) - apply(&mut g, vote(1, "a", "b", 3, 1)); - apply(&mut g, vote(2, "c", "d", 3, 1)); - - let n = g.idx_to_item.len(); - let (mut comps, isolates) = - connected_components_from_voted_pairs(n, g.voted_pairs.iter().copied()); + let mut scope = mk_scope(); + apply(&mut scope, vote(1, "a", "b", 3, 1)); + apply(&mut scope, vote(2, "c", "d", 3, 1)); + + let (_, idx_to_item) = item_index(&scope); + let (mut comps, isolates, _) = scope_components(&scope); assert!(isolates.is_empty()); - // Order-independent: sort components by their item names for stable assert. comps.sort_by_key(|c| { c.iter() - .map(|&i| g.idx_to_item[i].clone()) + .map(|&i| idx_to_item[i].clone()) .collect::>() }); assert_eq!(comps.len(), 2); - let comp0 = comps[0] - .iter() - .map(|&i| g.idx_to_item[i].as_str()) - .collect::>(); - let comp1 = comps[1] - .iter() - .map(|&i| g.idx_to_item[i].as_str()) - .collect::>(); - assert_eq!(comp0, vec!["a", "b"]); - assert_eq!(comp1, vec!["c", "d"]); } - /// A random spanning tree over 26 items needs only n−1 = 25 pairwise votes. - /// When each vote uses the "perfect" ratio (strength left : strength right = - /// (idx_left+1) : (idx_right+1)), rank centrality recovers the true order. - /// See `rank-eric.py` (Eric's demo of Negahban–Oh–Shah rank centrality). #[test] fn twenty_five_random_votes_perfect_ratios_sort_alphabet() { use rand::seq::SliceRandom; @@ -390,13 +403,13 @@ mod tests { let mut perm: Vec = (0..N).collect(); perm.shuffle(&mut rng); - let mut g = mk_group(); + let mut scope = mk_scope(); for k in 1..N { let i = *perm[..k].choose(&mut rng).unwrap(); let j = perm[k]; let (a, b) = (letters[i], letters[j]); apply( - &mut g, + &mut scope, vote( k as i64, &a.to_string(), @@ -407,34 +420,25 @@ mod tests { ); } - let ranked = ranked_items(&g); + let ranked = ranked_items(&scope); assert_eq!(ranked.len(), N); for (rank, item) in ranked.iter().enumerate() { let expected = char::from(b'a' + (N - 1 - rank) as u8); - assert_eq!( - item.item.as_str(), - expected.to_string(), - "rank {rank}: expected '{expected}', got '{}'", - item.item.as_str() - ); + assert_eq!(item.item.as_str(), expected.to_string()); } } #[test] fn subset_ranking_ranks_within_component_only() { - let mut g = mk_group(); - apply(&mut g, vote(1, "a", "b", 3, 1)); // a > b - apply(&mut g, vote(2, "c", "d", 1, 4)); // d > c - - let (comps, _) = connected_components_from_voted_pairs( - g.idx_to_item.len(), - g.voted_pairs.iter().copied(), - ); + let mut scope = mk_scope(); + apply(&mut scope, vote(1, "a", "b", 3, 1)); + apply(&mut scope, vote(2, "c", "d", 1, 4)); + + let (comps, _, _) = scope_components(&scope); assert_eq!(comps.len(), 2); - // Rank each component and ensure winner is first within that component. for comp in comps { - let ranked = ranked_items_subset(&g, &comp, 10000, 1e-8); + let ranked = ranked_items_subset(&scope, &comp, 10000, 1e-8); assert_eq!(ranked.len(), 2); let names = ranked.iter().map(|r| r.item.as_str()).collect::>(); if names.contains(&"a") { diff --git a/server/src/reducer.rs b/server/src/reducer.rs index 8c4c9f83635cbcbb037d680dbae2aa99cb2dc785..e6d8c8d2fe763f4928cfbd7c300c8d9768968a3f 100644 --- a/server/src/reducer.rs +++ b/server/src/reducer.rs @@ -4,6 +4,24 @@ use serde::{Deserialize, Serialize}; use crate::path_types::ItemId; +/// `(actor_uuid, min_item_id, max_item_id)` — one vote slot per human per pair. +pub type UuidVoteKey = (String, String, String); + +pub fn canonical_pair_ids(a: &ItemId, b: &ItemId) -> (String, String) { + let ak = a.as_str().to_string(); + let bk = b.as_str().to_string(); + if ak <= bk { + (ak, bk) + } else { + (bk, ak) + } +} + +pub fn uuid_vote_key(actor_uuid: &str, a: &ItemId, b: &ItemId) -> UuidVoteKey { + let (lo, hi) = canonical_pair_ids(a, b); + (actor_uuid.to_string(), lo, hi) +} + /// Parsed pairwise vote (internal representation). #[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] pub struct VoteData { @@ -44,118 +62,20 @@ impl VoteData { } } +/// Votes cast within one ranking scope (parent node). Edges and rankings are +/// derived on demand from [`Self::uuid_votes`]. #[derive(Debug, Clone, Default, Serialize, Deserialize)] -pub struct GroupState { - pub item_to_idx: HashMap, - pub idx_to_item: Vec, - pub edges: HashMap<(usize, usize), f64>, - pub voted_pairs: HashSet<(usize, usize)>, - /// Latest vote per `(actor_uuid, min_idx, max_idx)` — Sybil dedup anchor. - pub uuid_votes: HashMap<(String, usize, usize), VoteData>, +pub struct ScopeVotes { + pub uuid_votes: HashMap, pub recent_votes: Vec, } -impl GroupState { - pub fn new() -> Self { - Self { - item_to_idx: HashMap::new(), - idx_to_item: Vec::new(), - edges: HashMap::new(), - voted_pairs: HashSet::new(), - uuid_votes: HashMap::new(), - recent_votes: Vec::new(), - } - } - - fn ensure_item(&mut self, item: &ItemId) -> usize { - if let Some(&idx) = self.item_to_idx.get(item) { - return idx; - } - let idx = self.idx_to_item.len(); - self.idx_to_item.push(item.clone()); - self.item_to_idx.insert(item.clone(), idx); - idx - } - - fn add_edge_weight(&mut self, src: usize, dst: usize, w: f64) { - if w <= 0.0 { - return; - } - *self.edges.entry((src, dst)).or_insert(0.0) += w; - } - - fn subtract_edge_weight(&mut self, src: usize, dst: usize, w: f64) { - if w <= 0.0 { - return; - } - if let Some(entry) = self.edges.get_mut(&(src, dst)) { - *entry -= w; - if *entry <= 0.0 { - self.edges.remove(&(src, dst)); - } - } - } - - fn apply_weights(&mut self, vote: &VoteData, a_idx: usize, b_idx: usize) { - let w_a = vote.ratio_left as f64 * vote.trust_weight; - let w_b = vote.ratio_right as f64 * vote.trust_weight; - let (i, j) = if a_idx < b_idx { - (a_idx, b_idx) - } else { - (b_idx, a_idx) - }; - self.voted_pairs.insert((i, j)); - self.add_edge_weight(b_idx, a_idx, w_a); - self.add_edge_weight(a_idx, b_idx, w_b); - } - - fn rollback_weights(&mut self, vote: &VoteData) { - let a_idx = match self.item_to_idx.get(&vote.a) { - Some(&i) => i, - None => return, - }; - let b_idx = match self.item_to_idx.get(&vote.b) { - Some(&i) => i, - None => return, - }; - let w_a = vote.ratio_left as f64 * vote.trust_weight; - let w_b = vote.ratio_right as f64 * vote.trust_weight; - self.subtract_edge_weight(b_idx, a_idx, w_a); - self.subtract_edge_weight(a_idx, b_idx, w_b); - } - - /// Apply a validated vote, deduplicating by `actor_uuid` per unordered pair. +impl ScopeVotes { pub fn apply_vote(&mut self, vote: VoteData, actor_uuid: &str) { - let a_idx = self.ensure_item(&vote.a); - let b_idx = self.ensure_item(&vote.b); - let (i, j) = if a_idx < b_idx { - (a_idx, b_idx) - } else { - (b_idx, a_idx) - }; - - let dedupe_key = (actor_uuid.to_string(), i, j); - if let Some(old) = self.uuid_votes.get(&dedupe_key).cloned() { - self.rollback_weights(&old); - } - - self.apply_weights(&vote, a_idx, b_idx); - self.uuid_votes.insert(dedupe_key, vote.clone()); + let key = uuid_vote_key(actor_uuid, &vote.a, &vote.b); + self.uuid_votes.insert(key, vote.clone()); self.recent_votes.push(vote); } - - /// Rebuild edge weights from deduped uuid votes (load path — no rollback). - pub fn ingest_uuid_vote(&mut self, vote: VoteData, actor_uuid: &str) { - let a_idx = self.ensure_item(&vote.a); - let b_idx = self.ensure_item(&vote.b); - let (i, j) = if a_idx < b_idx { - (a_idx, b_idx) - } else { - (b_idx, a_idx) - }; - self.apply_weights(&vote, a_idx, b_idx); - self.uuid_votes.insert((actor_uuid.to_string(), i, j), vote); - } } /// Structured data imported from Reddit or elsewhere. @@ -179,7 +99,7 @@ pub struct NodeState { /// Ephemeral display view (Reddit title/author/etc.; not event-logged). pub data: Option, pub children: HashSet, - pub local_ranking: GroupState, + pub votes: ScopeVotes, } impl NodeState { @@ -242,7 +162,7 @@ impl GlobalTree { if let Some(node) = self.nodes.get_mut(parent) { node.children.insert(vote.a.clone()); node.children.insert(vote.b.clone()); - node.local_ranking.apply_vote(vote, actor_uuid); + node.votes.apply_vote(vote, actor_uuid); } } @@ -276,6 +196,7 @@ impl GlobalTree { #[cfg(test)] mod tests { use super::*; + use crate::ranking::edge_weight_sum; fn vote(ts: i64, a: &str, b: &str, l: i32, r: i32, pseudonym: &str) -> VoteData { VoteData { @@ -310,26 +231,23 @@ mod tests { #[test] fn same_uuid_replaces_prior_vote_on_pair() { - let mut g = GroupState::new(); + let mut scope = ScopeVotes::default(); let uuid = "u1"; - g.apply_vote(vote(1, "a", "b", 2, 1, "alice"), uuid); - let first_total: f64 = g.edges.values().sum(); - assert_eq!(first_total, 3.0); + scope.apply_vote(vote(1, "a", "b", 2, 1, "alice"), uuid); + assert_eq!(edge_weight_sum(&scope), 3.0); - g.apply_vote(vote(2, "a", "b", 0, 1, "bob"), uuid); - let second_total: f64 = g.edges.values().sum(); - assert_eq!(second_total, 1.0); - assert_eq!(g.uuid_votes.len(), 1); + scope.apply_vote(vote(2, "a", "b", 0, 1, "bob"), uuid); + assert_eq!(edge_weight_sum(&scope), 1.0); + assert_eq!(scope.uuid_votes.len(), 1); } #[test] fn different_uuids_both_count() { - let mut g = GroupState::new(); - g.apply_vote(vote(1, "a", "b", 2, 1, "alice"), "u1"); - g.apply_vote(vote(2, "a", "b", 0, 1, "bob"), "u2"); - let total: f64 = g.edges.values().sum(); - assert_eq!(total, 4.0); - assert_eq!(g.uuid_votes.len(), 2); + let mut scope = ScopeVotes::default(); + scope.apply_vote(vote(1, "a", "b", 2, 1, "alice"), "u1"); + scope.apply_vote(vote(2, "a", "b", 0, 1, "bob"), "u2"); + assert_eq!(edge_weight_sum(&scope), 4.0); + assert_eq!(scope.uuid_votes.len(), 2); } #[test] diff --git a/server/src/state.rs b/server/src/state.rs index 8dabc93e95c39cbe18d69596dee79683b384ef86..dcb82ff4beeaac8f0820de0e0131ca1f6bd81dcc 100644 --- a/server/src/state.rs +++ b/server/src/state.rs @@ -269,7 +269,7 @@ mod tests { use super::{normalize_scope, parse_item_param, AppConfig, AppState}; use crate::{ event_log::EventLog, events::Event, path_types::ItemId, projection_apply, - projection_store::ProjectionStore, reducer::EntityData, + projection_store::ProjectionStore, ranking::edge_weight_sum, reducer::EntityData, }; fn event_record(seq: u64, event: Event) -> crate::events::EventRecord { @@ -442,7 +442,7 @@ mod tests { assert_eq!(projection_store.last_applied_event_count().unwrap(), 1); let first = projection_store.scope_tree(&ItemId::root()).unwrap(); let first_root = first.get(&ItemId::root()).unwrap(); - let first_edge_total: f64 = first_root.local_ranking.edges.values().sum(); + let first_edge_total = edge_weight_sum(&first_root.votes); assert_eq!(first_edge_total, 3.0); super::catch_up_projection(&log, &projection_store) @@ -451,7 +451,7 @@ mod tests { assert_eq!(projection_store.last_applied_event_count().unwrap(), 1); let second = projection_store.scope_tree(&ItemId::root()).unwrap(); let second_root = second.get(&ItemId::root()).unwrap(); - let second_edge_total: f64 = second_root.local_ranking.edges.values().sum(); + let second_edge_total = edge_weight_sum(&second_root.votes); assert_eq!(second_edge_total, first_edge_total); } @@ -529,9 +529,8 @@ mod tests { let root = projected.get(&ItemId::root()).unwrap(); assert!(root.children.contains(&ItemId::parse("alpha").unwrap())); assert!(root.children.contains(&ItemId::parse("beta").unwrap())); - assert_eq!(root.local_ranking.idx_to_item.len(), 2); - let edge_total: f64 = root.local_ranking.edges.values().sum(); - assert_eq!(edge_total, 3.0); + assert_eq!(crate::ranking::ranked_items(&root.votes).len(), 2); + assert_eq!(edge_weight_sum(&root.votes), 3.0); } #[tokio::test] @@ -620,7 +619,7 @@ mod tests { let root = tree.get(&ItemId::root()).unwrap(); assert!(root.children.contains(&ItemId::parse("beta").unwrap())); assert!(root.children.contains(&ItemId::parse("gamma").unwrap())); - assert_eq!(root.local_ranking.idx_to_item.len(), 3); + assert_eq!(crate::ranking::ranked_items(&root.votes).len(), 3); } #[test] diff --git a/server/src/storage_schema.rs b/server/src/storage_schema.rs index 67c9f4ab2e1865f8da81b6735dd2a05b87e5e366..fe2671b876f3df77fdfd402dbc845ff1eb3512cd 100644 --- a/server/src/storage_schema.rs +++ b/server/src/storage_schema.rs @@ -3,86 +3,53 @@ //! //! Votes are stored as deduped `uuid_votes` entries plus an append-only //! `recent_votes` audit list. Edge weights for rank centrality are derived -//! from `uuid_votes` on read, not incrementally merged in RocksDB. +//! from `uuid_votes` on read, not stored in RocksDB. -use std::collections::{HashSet}; +use std::collections::HashSet; use durable::{Batch, Db, Durable, Leaf, List, Map}; use crate::{ path_types::ItemId, - reducer::{EntityData, GroupState, NodeState, VoteData}, + reducer::{EntityData, NodeState, ScopeVotes, VoteData, UuidVoteKey, uuid_vote_key}, storage_dto::{ decode_entity_data, decode_vote, encode_entity_data, encode_vote, parse_stored_id, StoredEntityDataV1, StoredVoteV1, }, }; -/// `(actor_uuid, min_item_id, max_item_id)` — one vote slot per human per pair. -pub type UuidVoteKey = (String, String, String); - /// One node in the fractal tree, exploded into precisely-updatable collections. #[derive(Durable)] #[allow(dead_code)] pub struct NodeSchema { - /// Presence marker (a node "exists" once ensured/voted/imported). pub present: Leaf, - /// Domain-specific derived view (Reddit title/author/…); absent => None. pub data: Leaf, - /// Child ids (a set; value is always `true`). pub children: Map>, - /// Latest vote per actor per unordered pair; edges are derived from this on read. pub uuid_votes: Map>, - /// Recent votes, append-only oldest-first (cap applied on read). pub recent_votes: List>, - /// When ephemeral Reddit display content was last fetched (ms); absent after eviction. pub fetched_at: Leaf, } -/// The single database root: nodes, identity maps, view counts, and metadata. #[derive(Durable)] #[allow(dead_code)] pub struct Store { pub nodes: Map, - /// Global pseudonym → actor UUID (Sybil dedup anchor). pub pseudonyms: Map>, pub proj_meta: Map>, pub view_counts: Map>, pub view_meta: Map>, } -/// Max recent votes returned when loading a node (query-time cap only). pub const RECENT_VOTES_CAP: u64 = 200; fn id_key(id: &ItemId) -> String { id.as_str().to_string() } -fn pair_keys(a: &ItemId, b: &ItemId) -> (String, String) { - let ak = id_key(a); - let bk = id_key(b); - if ak <= bk { - (ak, bk) - } else { - (bk, ak) - } -} - -pub fn uuid_vote_key(actor_uuid: &str, a: &ItemId, b: &ItemId) -> UuidVoteKey { - let (lo, hi) = pair_keys(a, b); - (actor_uuid.to_string(), lo, hi) -} - -/// Path to a node by id. pub fn node(id: &ItemId) -> durable::Path { Store::root().nodes().key(&id_key(id)) } -// --------------------------------------------------------------------------- -// Reconstruction (durable -> in-memory) -// --------------------------------------------------------------------------- - -/// Reconstruct a node's in-memory state, or `None` if the node does not exist. pub fn load_node_state(db: &Db, id: &ItemId) -> durable::Result> { let np = node(id); let present = np.present().get(db)?.unwrap_or(false); @@ -100,47 +67,39 @@ pub fn load_node_state(db: &Db, id: &ItemId) -> durable::Result) -> durable::Result { - let mut group = GroupState::new(); +fn load_scope_votes(db: &Db, np: &durable::Path) -> durable::Result { + let mut votes = ScopeVotes::default(); for (key, stored) in np.uuid_votes().iter(db)? { - let (actor_uuid, _lo, _hi) = key; let vote = decode_vote(stored).map_err(durable::Error::Deserialize)?; - group.ingest_uuid_vote(vote, &actor_uuid); + votes.uuid_votes.insert(key, vote); } let stored = np.recent_votes().iter(db)?; let cap = RECENT_VOTES_CAP as usize; let start = stored.len().saturating_sub(cap); - group.recent_votes = stored[start..] + votes.recent_votes = stored[start..] .iter() .map(|s| decode_vote(s.clone()).map_err(durable::Error::Deserialize)) .collect::, _>>()?; - Ok(group) + Ok(votes) } fn parse_storage_id(s: &str) -> durable::Result { parse_stored_id(s).map_err(durable::Error::Deserialize) } -// --------------------------------------------------------------------------- -// Write helpers (event -> reified point updates on a batch) -// --------------------------------------------------------------------------- - -/// Wire a node and its ancestors into the tree exactly like -/// [`crate::reducer::GlobalTree::ensure_path`]: set presence and parent→child -/// links along the canonical breadcrumb path. pub fn ensure_path_writes(batch: &mut Batch, id: &ItemId) { let root = ItemId::root(); batch.write(node(&root).present().set(&true)); @@ -161,7 +120,6 @@ pub fn ensure_path_writes(batch: &mut Batch, id: &ItemId) { } } -/// Reified writes for a validated vote under `parent`. pub fn vote_writes( batch: &mut Batch, parent: &ItemId, @@ -182,14 +140,12 @@ pub fn vote_writes( Ok(()) } -/// Reified writes for ephemeral Reddit display content (not event-logged). pub fn entity_content_writes(batch: &mut Batch, id: &ItemId, view: &EntityData, fetched_at: i64) { ensure_path_writes(batch, id); batch.write(node(id).data().set(&encode_entity_data(view))); batch.write(node(id).fetched_at().set(&fetched_at)); } -/// Clear cached display content for one node (structure/votes are untouched). pub fn entity_content_clear_writes(batch: &mut Batch, id: &ItemId) { batch.write(node(id).data().delete()); batch.write(node(id).fetched_at().delete()); @@ -199,6 +155,7 @@ pub fn entity_content_clear_writes(batch: &mut Batch, id: &ItemId) { mod tests { use super::*; use crate::identity::{seed_default_pseudonym, DEFAULT_ACTOR_UUID, DEFAULT_PSEUDONYM}; + use crate::ranking::edge_weight_sum; fn sample_vote(ts: i64, a: &str, b: &str, l: i32, r: i32) -> VoteData { VoteData { @@ -213,7 +170,7 @@ mod tests { } #[test] - fn vote_roundtrip_reconstructs_group_state() { + fn vote_roundtrip_reconstructs_ranking() { let dir = tempfile::tempdir().unwrap(); let db = Db::open(dir.path()).unwrap(); seed_default_pseudonym(&db).unwrap(); @@ -225,11 +182,8 @@ mod tests { batch.commit().unwrap(); let node_state = load_node_state(&db, &parent).unwrap().unwrap(); - let g = &node_state.local_ranking; - assert_eq!(g.idx_to_item.len(), 2); - let edge_total: f64 = g.edges.values().sum(); - assert_eq!(edge_total, 3.0); - assert_eq!(g.recent_votes.len(), 1); + assert_eq!(edge_weight_sum(&node_state.votes), 3.0); + assert_eq!(node_state.votes.recent_votes.len(), 1); assert!(node_state.children.contains(&ItemId::opaque("alpha"))); assert!(node_state.children.contains(&ItemId::opaque("beta"))); } @@ -263,10 +217,9 @@ mod tests { .unwrap(); batch.commit().unwrap(); - let g = &load_node_state(&db, &parent).unwrap().unwrap().local_ranking; - let edge_total: f64 = g.edges.values().sum(); - assert_eq!(edge_total, 1.0); - assert_eq!(g.uuid_votes.len(), 1); + let votes = &load_node_state(&db, &parent).unwrap().unwrap().votes; + assert_eq!(edge_weight_sum(votes), 1.0); + assert_eq!(votes.uuid_votes.len(), 1); } #[test] @@ -293,15 +246,8 @@ mod tests { ); let node_state = load_node_state(&db, &parent).unwrap().unwrap(); - assert_eq!(node_state.local_ranking.recent_votes.len(), RECENT_VOTES_CAP as usize); - assert_eq!( - node_state - .local_ranking - .recent_votes - .first() - .map(|v| v.ts), - Some(10) - ); + assert_eq!(node_state.votes.recent_votes.len(), RECENT_VOTES_CAP as usize); + assert_eq!(node_state.votes.recent_votes.first().map(|v| v.ts), Some(10)); } #[test] diff --git a/server/tests/integration_ui.rs b/server/tests/integration_ui.rs index cc7a16d756673b95ba336f2d6130eaf40908cc60..40e4cfe1eec36dad29a075fb01aefb4dffd856d0 100644 --- a/server/tests/integration_ui.rs +++ b/server/tests/integration_ui.rs @@ -132,7 +132,7 @@ async fn post_ui_record_vote_morphs_ranking_and_persists() { let state = create_app_state(cfg).await; let tree = state.scope_tree(&ItemId::root()).unwrap(); let root = tree.get(&ItemId::root()).expect("root node after replay"); - let ranked = sorter2_server::ranking::ranked_items(&root.local_ranking); + let ranked = sorter2_server::ranking::ranked_items(&root.votes); assert_eq!(ranked.len(), 2); assert_eq!(ranked[0].item.as_str(), "alpha"); } Side B — contributor: tommy-mor Side B — commit message: [15e1037a] url stuff Side B — unified diff (full patch): diff --git a/server/src/url_rules/graph.rs b/server/src/url_rules/graph.rs new file mode 100644 index 0000000000000000000000000000000000000000..f7ac0f9a551a1727cb2f9294778c283b9885b147 --- /dev/null +++ b/server/src/url_rules/graph.rs @@ -0,0 +1,831 @@ +//! Semantic URL graph: DFA traversal on host + path, query in context, generic fallback. + +use std::collections::HashMap; +use std::sync::OnceLock; + +use url::Url; + +use super::graph_builder::GraphBuilder; +use super::parse::{normalize_match_host, strip_tracking_query, UrlParts}; + +#[derive(Debug, Clone, Default)] +pub struct Context { + pub vars: HashMap, + pub query: HashMap, +} + +pub type CanonicalFn = fn(&Context) -> Option; + +#[derive(Clone, Copy)] +pub enum EdgePattern { + Literal(&'static str), + Variable(&'static str), + /// Absorb any trailing segment without leaving this node (e.g. post title slug). + AbsorbAny, + /// Absorb segment when `cond(seg)` (e.g. subreddit listing suffix). + AbsorbIf(fn(&str) -> bool), +} + +pub struct Edge { + pub pattern: EdgePattern, + pub target: &'static str, +} + +pub struct Node { + pub edges: Vec, + pub canonical: CanonicalFn, + pub parent: Option<&'static str>, +} + +impl Node { + pub(crate) fn empty() -> Self { + Self { + edges: Vec::new(), + canonical: |_| None, + parent: None, + } + } +} + +pub struct Graph { + pub nodes: HashMap<&'static str, Node>, +} + +static GRAPH: OnceLock = OnceLock::new(); + +pub fn graph() -> &'static Graph { + GRAPH.get_or_init(build_graph) +} + +impl Graph { + pub fn resolve_canonical(&self, parts: &UrlParts) -> Option { + let mut query = parts.query.clone(); + strip_tracking_query(&mut query); + let mut ctx = Context { + vars: HashMap::new(), + query, + }; + + if let Some(node_id) = self.traverse(parts, &mut ctx) { + if let Some(canon) = (self.nodes.get(node_id)?.canonical)(&ctx) { + return Some(canon); + } + } + Some(generic_canonical(parts)) + } + + pub fn breadcrumbs(&self, parts: &UrlParts) -> Vec { + let mut query = parts.query.clone(); + strip_tracking_query(&mut query); + let mut ctx = Context { + vars: HashMap::new(), + query, + }; + + if let Some(mut node_id) = self.traverse(parts, &mut ctx) { + let mut paths = Vec::new(); + loop { + let node = match self.nodes.get(node_id) { + Some(n) => n, + None => break, + }; + if let Some(url) = (node.canonical)(&ctx) { + if paths.last() != Some(&url) { + paths.push(url); + } + } + match node.parent { + Some(p) => node_id = p, + None => break, + } + } + paths.reverse(); + if !paths.is_empty() { + return paths; + } + } + generic_breadcrumbs(parts) + } + + fn traverse(&self, parts: &UrlParts, ctx: &mut Context) -> Option<&'static str> { + let host = parts.match_host(); + let mut node_id = match host.as_str() { + "reddit.com" => "reddit_root", + "youtube.com" => "youtube_root", + "youtu.be" => "youtu_be_entry", + _ => return None, + }; + + let segs: Vec<&str> = parts.path_segments.iter().map(String::as_str).collect(); + let mut i = 0; + while i < segs.len() { + let seg = segs[i]; + match self.follow_edge(node_id, seg, ctx) { + Ok(next) => { + node_id = next; + i += 1; + } + Err(()) => { + if self.try_absorb(node_id, seg) { + i += 1; + continue; + } + return None; + } + } + } + Some(node_id) + } + + fn follow_edge( + &self, + node_id: &'static str, + seg: &str, + ctx: &mut Context, + ) -> Result<&'static str, ()> { + let node = self.nodes.get(node_id).ok_or(())?; + for edge in &node.edges { + match edge.pattern { + EdgePattern::Literal(lit) if lit == seg => return Ok(edge.target), + EdgePattern::Variable(name) => { + ctx.vars.insert(name.to_string(), seg.to_string()); + return Ok(edge.target); + } + EdgePattern::AbsorbAny + | EdgePattern::AbsorbIf(_) + | EdgePattern::Literal(_) + | EdgePattern::Variable(_) => {} + } + } + Err(()) + } + + fn try_absorb(&self, node_id: &'static str, seg: &str) -> bool { + let node = match self.nodes.get(node_id) { + Some(n) => n, + None => return false, + }; + for edge in &node.edges { + match edge.pattern { + EdgePattern::AbsorbAny => return true, + EdgePattern::AbsorbIf(cond) if cond(seg) => return true, + EdgePattern::AbsorbIf(_) | EdgePattern::Literal(_) | EdgePattern::Variable(_) => {} + } + } + false + } + + /// Test hook: terminal graph node and captured context after traversal. + #[cfg(test)] + pub fn traverse_terminal(&self, parts: &UrlParts) -> Option<(&'static str, Context)> { + let mut query = parts.query.clone(); + strip_tracking_query(&mut query); + let mut ctx = Context { + vars: HashMap::new(), + query, + }; + let node = self.traverse(parts, &mut ctx)?; + Some((node, ctx)) + } +} + +fn is_reddit_listing_suffix(seg: &str) -> bool { + matches!(seg, "hot" | "top" | "new" | "rising" | "controversial") +} + +/// Percent-encode a path or query fragment so `&`, `?`, etc. cannot break URL structure. +fn enc(s: &str) -> String { + urlencoding::encode(s).into_owned() +} + +// --- Canonical formatters --- + +fn canon_reddit_root(_: &Context) -> Option { + Some("https://reddit.com".to_string()) +} + +fn canon_reddit_r_hub(_: &Context) -> Option { + Some("https://reddit.com/r".to_string()) +} + +fn canon_reddit_subreddit(ctx: &Context) -> Option { + let sub = ctx.vars.get("subreddit")?; + Some(format!( + "https://reddit.com/r/{}", + enc(&sub.to_ascii_lowercase()) + )) +} + +fn canon_reddit_post(ctx: &Context) -> Option { + let sub = ctx.vars.get("subreddit")?.to_ascii_lowercase(); + let id = ctx.vars.get("post_id")?; + Some(format!( + "https://reddit.com/r/{}/comments/{}", + enc(&sub), + enc(id) + )) +} + +fn canon_youtube_root(_: &Context) -> Option { + Some("https://youtube.com".to_string()) +} + +fn canon_youtube_watch(ctx: &Context) -> Option { + let v = ctx + .query + .get("v") + .or_else(|| ctx.vars.get("video_id"))?; + Some(format!("https://youtube.com/watch?v={}", enc(v))) +} + +fn canon_youtu_be(ctx: &Context) -> Option { + let v = ctx.vars.get("vid_id")?; + Some(format!("https://youtube.com/watch?v={}", enc(v))) +} + +pub fn build_graph() -> Graph { + GraphBuilder::new() + .node("reddit_root") + .canonical(canon_reddit_root) + .edge(EdgePattern::Literal("r"), "reddit_r_hub") + .node("reddit_r_hub") + .parent("reddit_root") + .canonical(canon_reddit_r_hub) + .edge(EdgePattern::Variable("subreddit"), "reddit_subreddit") + .node("reddit_subreddit") + .parent("reddit_r_hub") + .canonical(canon_reddit_subreddit) + .edge( + EdgePattern::AbsorbIf(is_reddit_listing_suffix), + "reddit_subreddit", + ) + .edge(EdgePattern::Literal("comments"), "reddit_comments_gate") + .node("reddit_comments_gate") + .parent("reddit_subreddit") + .canonical(canon_reddit_subreddit) + .edge(EdgePattern::Variable("post_id"), "reddit_post") + .node("reddit_post") + .parent("reddit_subreddit") + .canonical(canon_reddit_post) + .edge(EdgePattern::AbsorbAny, "reddit_post") + .node("youtube_root") + .canonical(canon_youtube_root) + .edge(EdgePattern::Literal("watch"), "youtube_watch") + .edge(EdgePattern::Literal("shorts"), "youtube_shorts_gate") + .node("youtube_watch") + .parent("youtube_root") + .canonical(canon_youtube_watch) + .node("youtube_shorts_gate") + .parent("youtube_root") + .canonical(canon_youtube_root) + .edge(EdgePattern::Variable("video_id"), "youtube_watch") + .node("youtu_be_entry") + .canonical(canon_youtube_root) + .edge(EdgePattern::Variable("vid_id"), "youtu_be_video") + .node("youtu_be_video") + .parent("youtube_root") + .canonical(canon_youtu_be) + .build() +} + +// --- Generic internet fallback --- + +pub fn generic_canonical(parts: &UrlParts) -> String { + let host = normalize_match_host(&parts.host); + let path_segments: Vec = parts.path_segments.clone(); + let mut query = parts.query.clone(); + strip_tracking_query(&mut query); + + let mut url = if path_segments.is_empty() { + Url::parse(&format!("https://{host}")) + .unwrap_or_else(|_| Url::parse("https://invalid").unwrap()) + } else { + let path = format!("/{}", path_segments.join("/")); + Url::parse(&format!("https://{host}{path}")) + .unwrap_or_else(|_| Url::parse("https://invalid").unwrap()) + }; + + if !query.is_empty() { + let mut pairs: Vec<_> = query.iter().collect(); + pairs.sort_by(|a, b| a.0.cmp(b.0)); + url.query_pairs_mut().clear(); + for (k, v) in pairs { + url.query_pairs_mut().append_pair(k, v); + } + } + + let mut s = url.to_string(); + if path_segments.is_empty() { + s = s.trim_end_matches('/').to_string(); + } + s +} + +pub fn generic_breadcrumbs(parts: &UrlParts) -> Vec { + let host = normalize_match_host(&parts.host); + let n = parts.path_segments.len(); + let mut out = Vec::new(); + + let base = generic_canonical(&UrlParts { + scheme: "https".to_string(), + host: host.clone(), + path_segments: vec![], + query: HashMap::new(), + }); + out.push(base); + + for i in 0..n { + let segs: Vec = parts.path_segments[..=i].to_vec(); + let url = generic_canonical(&UrlParts { + scheme: "https".to_string(), + host: host.clone(), + path_segments: segs, + query: HashMap::new(), + }); + if out.last() != Some(&url) { + out.push(url); + } + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::url_rules::parse::test_parts; + + fn g() -> &'static Graph { + graph() + } + + fn canon(parts: &UrlParts) -> String { + g().resolve_canonical(parts).unwrap() + } + + fn crumbs(parts: &UrlParts) -> Vec { + g().breadcrumbs(parts) + } + + fn terminal(parts: &UrlParts) -> Option<&'static str> { + g().traverse_terminal(parts).map(|(n, _)| n) + } + + fn vars(parts: &UrlParts) -> HashMap { + g().traverse_terminal(parts) + .map(|(_, c)| c.vars) + .unwrap_or_default() + } + + #[test] + fn youtu_be_malicious_segment_encoded_not_injected() { + let p = test_parts("youtu.be", &["abc&t=1"], &[]); + assert_eq!(canon(&p), "https://youtube.com/watch?v=abc%26t%3D1"); + assert!(!canon(&p).contains("abc&t=1")); + } + + #[test] + fn youtube_query_v_encoded() { + let p = test_parts("youtube.com", &["watch"], &[("v", "a&b=c")]); + assert_eq!(canon(&p), "https://youtube.com/watch?v=a%26b%3Dc"); + } + + #[test] + fn absorb_patterns_live_on_edges_not_in_engine() { + let g = build_graph(); + let sub = g.nodes.get("reddit_subreddit").unwrap(); + assert!(sub + .edges + .iter() + .any(|e| matches!(e.pattern, EdgePattern::AbsorbIf(_)))); + let post = g.nodes.get("reddit_post").unwrap(); + assert!(post + .edges + .iter() + .any(|e| matches!(e.pattern, EdgePattern::AbsorbAny))); + } + + #[test] + fn builder_rejects_missing_parent() { + let result = std::panic::catch_unwind(|| { + GraphBuilder::new() + .node("orphan") + .parent("nonexistent_parent") + .build(); + }); + assert!(result.is_err()); + } + + #[test] + fn generic_canonical_forces_https_and_strips_www() { + let p = test_parts("www.example.com", &["blog", "post"], &[]); + assert_eq!(canon(&p), "https://example.com/blog/post"); + } + + #[test] + fn generic_canonical_sorts_query_keys() { + let p = test_parts("example.com", &["search"], &[("q", "rust"), ("page", "2")]); + assert_eq!(canon(&p), "https://example.com/search?page=2&q=rust"); + } + + #[test] + fn generic_canonical_strips_tracking_from_query() { + let p = test_parts( + "news.ycombinator.com", + &["item"], + &[("id", "1"), ("utm_medium", "social")], + ); + assert_eq!(canon(&p), "https://news.ycombinator.com/item?id=1"); + } + + #[test] + fn generic_breadcrumbs_cumulative_path() { + let p = test_parts("paulgraham.com", &["articles", "lisp.html"], &[]); + assert_eq!( + crumbs(&p), + vec![ + "https://paulgraham.com", + "https://paulgraham.com/articles", + "https://paulgraham.com/articles/lisp.html" + ] + ); + } + + #[test] + fn generic_breadcrumbs_domain_only() { + let p = test_parts("example.com", &[], &[]); + assert_eq!(crumbs(&p), vec!["https://example.com"]); + } + + #[test] + fn unknown_host_uses_generic_not_graph() { + let p = test_parts("hackernews.com", &["item", "123"], &[]); + assert_eq!(terminal(&p), None); + assert_eq!(canon(&p), "https://hackernews.com/item/123"); + } + + #[test] + fn traverse_captures_subreddit_variable() { + let p = test_parts("reddit.com", &["r", "Rust"], &[]); + assert_eq!(terminal(&p), Some("reddit_subreddit")); + assert_eq!(vars(&p).get("subreddit").map(String::as_str), Some("Rust")); + } + + #[test] + fn traverse_captures_post_id() { + let p = test_parts("reddit.com", &["r", "aww", "comments", "abc123"], &[]); + assert_eq!(terminal(&p), Some("reddit_post")); + assert_eq!(vars(&p).get("post_id").map(String::as_str), Some("abc123")); + } + + #[test] + fn traverse_absorbs_listing_suffix_stays_on_subreddit() { + let p = test_parts("reddit.com", &["r", "rust", "hot"], &[]); + assert_eq!(terminal(&p), Some("reddit_subreddit")); + assert_eq!(canon(&p), "https://reddit.com/r/rust"); + } + + #[test] + fn traverse_absorbs_all_listing_suffixes() { + for suffix in ["hot", "top", "new", "rising", "controversial"] { + let p = test_parts("reddit.com", &["r", "test", suffix], &[]); + assert_eq!(terminal(&p), Some("reddit_subreddit"), "suffix {suffix}"); + assert_eq!(canon(&p), "https://reddit.com/r/test", "suffix {suffix}"); + } + } + + #[test] + fn traverse_absorbs_post_title_slug() { + let p = test_parts( + "reddit.com", + &["r", "rust", "comments", "aaa", "my_great_post_title"], + &[], + ); + assert_eq!(terminal(&p), Some("reddit_post")); + assert_eq!(canon(&p), "https://reddit.com/r/rust/comments/aaa"); + } + + #[test] + fn traverse_unknown_segment_falls_back_to_generic() { + let p = test_parts("reddit.com", &["r", "rust", "wiki", "faq"], &[]); + assert_eq!(terminal(&p), None); + assert_eq!(canon(&p), "https://reddit.com/r/rust/wiki/faq"); + } + + #[test] + fn traverse_youtube_watch_requires_v_in_query() { + let p = test_parts("youtube.com", &["watch"], &[("v", "xyz")]); + assert_eq!(terminal(&p), Some("youtube_watch")); + } + + #[test] + fn traverse_youtu_be_captures_vid_id() { + let p = test_parts("youtu.be", &["dQw4w9WgXcQ"], &[]); + assert_eq!(terminal(&p), Some("youtu_be_video")); + assert_eq!(vars(&p).get("vid_id").map(String::as_str), Some("dQw4w9WgXcQ")); + } + + #[test] + fn traverse_shorts_sets_video_id_var() { + let p = test_parts("youtube.com", &["shorts", "abc99"], &[]); + assert_eq!(terminal(&p), Some("youtube_watch")); + assert_eq!(vars(&p).get("video_id").map(String::as_str), Some("abc99")); + } + + #[test] + fn reddit_domain_canonical() { + let p = test_parts("reddit.com", &[], &[]); + assert_eq!(canon(&p), "https://reddit.com"); + } + + #[test] + fn reddit_r_hub_canonical() { + let p = test_parts("reddit.com", &["r"], &[]); + assert_eq!(terminal(&p), Some("reddit_r_hub")); + assert_eq!(canon(&p), "https://reddit.com/r"); + } + + #[test] + fn reddit_subreddit_lowercases_name() { + let p = test_parts("reddit.com", &["r", "AmITheAsshole"], &[]); + assert_eq!(canon(&p), "https://reddit.com/r/amitheasshole"); + } + + #[test] + fn reddit_host_aliases_old_new_www() { + for host in ["old.reddit.com", "new.reddit.com", "www.reddit.com"] { + let p = test_parts(host, &["r", "rust"], &[]); + assert_eq!(canon(&p), "https://reddit.com/r/rust", "host {host}"); + } + } + + #[test] + fn reddit_post_strips_slug_and_query() { + let p = test_parts( + "old.reddit.com", + &["r", "Rust", "comments", "1abc", "title_slug_here"], + &[("sort", "new")], + ); + assert_eq!(canon(&p), "https://reddit.com/r/rust/comments/1abc"); + } + + #[test] + fn reddit_post_multiple_slugs_absorbed() { + let p = test_parts( + "reddit.com", + &["r", "x", "comments", "id1", "slug1", "extra"], + &[], + ); + assert_eq!(canon(&p), "https://reddit.com/r/x/comments/id1"); + } + + #[test] + fn reddit_listing_with_query_only() { + let p = test_parts("www.reddit.com", &["r", "programming"], &[("sort", "top")]); + assert_eq!(canon(&p), "https://reddit.com/r/programming"); + } + + #[test] + fn reddit_subreddit_breadcrumbs_include_r_hub() { + let p = test_parts("reddit.com", &["r", "movies"], &[]); + assert_eq!( + crumbs(&p), + vec![ + "https://reddit.com", + "https://reddit.com/r", + "https://reddit.com/r/movies" + ] + ); + } + + #[test] + fn reddit_post_breadcrumbs_skip_comments_node() { + let p = test_parts("reddit.com", &["r", "aww", "comments", "1trnvdl"], &[]); + let c = crumbs(&p); + assert!(!c.iter().any(|u| u.ends_with("/comments"))); + assert_eq!( + c.last().map(String::as_str), + Some("https://reddit.com/r/aww/comments/1trnvdl") + ); + assert!(c.contains(&"https://reddit.com/r/aww".to_string())); + } + + #[test] + fn reddit_post_parent_is_subreddit_not_comments() { + let p = test_parts("reddit.com", &["r", "aww", "comments", "1trnvdl"], &[]); + let c = crumbs(&p); + let parent = c.get(c.len() - 2).unwrap(); + assert_eq!(parent, "https://reddit.com/r/aww"); + } + + #[test] + fn reddit_domain_parent_is_none_in_breadcrumb_chain() { + let p = test_parts("reddit.com", &[], &[]); + assert_eq!(crumbs(&p), vec!["https://reddit.com"]); + } + + #[test] + fn youtube_watch_canonical_uses_v_only() { + let p = test_parts("youtube.com", &["watch"], &[("v", "abc"), ("t", "99")]); + assert_eq!(canon(&p), "https://youtube.com/watch?v=abc"); + } + + #[test] + fn youtube_query_order_independent() { + let a = test_parts("youtube.com", &["watch"], &[("v", "abc"), ("t", "4")]); + let b = test_parts("youtube.com", &["watch"], &[("t", "4"), ("v", "abc")]); + assert_eq!(canon(&a), canon(&b)); + } + + #[test] + fn youtube_host_aliases() { + for host in ["www.youtube.com", "m.youtube.com"] { + let p = test_parts(host, &["watch"], &[("v", "x")]); + assert_eq!(canon(&p), "https://youtube.com/watch?v=x", "host {host}"); + } + } + + #[test] + fn youtube_shorts_canonical_matches_watch() { + let shorts = test_parts("youtube.com", &["shorts", "vid123"], &[]); + let watch = test_parts("youtube.com", &["watch"], &[("v", "vid123")]); + assert_eq!(canon(&shorts), canon(&watch)); + assert_eq!(canon(&shorts), "https://youtube.com/watch?v=vid123"); + } + + #[test] + fn youtu_be_matches_youtube_watch() { + let be = test_parts("youtu.be", &["dQw4w9WgXcQ"], &[]); + let watch = test_parts("youtube.com", &["watch"], &[("v", "dQw4w9WgXcQ")]); + assert_eq!(canon(&be), canon(&watch)); + } + + #[test] + fn youtube_breadcrumbs_domain_then_watch() { + let p = test_parts("youtube.com", &["watch"], &[("v", "abc")]); + assert_eq!( + crumbs(&p), + vec!["https://youtube.com", "https://youtube.com/watch?v=abc"] + ); + } + + #[test] + fn youtu_be_breadcrumbs_include_youtube_domain() { + let p = test_parts("youtu.be", &["abc"], &[]); + let c = crumbs(&p); + assert_eq!(c.first().map(String::as_str), Some("https://youtube.com")); + assert_eq!( + c.last().map(String::as_str), + Some("https://youtube.com/watch?v=abc") + ); + } + + #[test] + fn parsed_urls_match_hand_built_parts() { + let raw = "https://www.reddit.com/r/rust/comments/aaa/title/?utm=x"; + let parsed = UrlParts::parse(raw).unwrap(); + let hand = test_parts( + "www.reddit.com", + &["r", "rust", "comments", "aaa", "title"], + &[("utm", "x")], + ); + assert_eq!(canon(&parsed), canon(&hand)); + } + + #[test] + fn equivalence_cluster_youtube_formats() { + let urls = [ + "https://youtu.be/abc123", + "https://www.youtube.com/watch?v=abc123", + "https://youtube.com/watch?v=abc123&t=1", + "https://m.youtube.com/watch?t=1&v=abc123", + ]; + let canonical: Vec<_> = urls + .iter() + .map(|u| canon(&UrlParts::parse(u).unwrap())) + .collect(); + assert!(canonical.iter().all(|c| *c == "https://youtube.com/watch?v=abc123")); + } + + #[test] + fn equivalence_cluster_reddit_post_formats() { + let urls = [ + "https://old.reddit.com/r/Rust/comments/aaa/slug/", + "reddit.com/r/rust/comments/aaa/other_slug", + "https://reddit.com/r/RUST/comments/aaa", + ]; + let canonical: Vec<_> = urls + .iter() + .map(|u| canon(&UrlParts::parse(u).unwrap())) + .collect(); + assert!( + canonical + .iter() + .all(|c| *c == "https://reddit.com/r/rust/comments/aaa") + ); + } + + #[test] + fn graph_nodes_all_have_valid_parent_links() { + let g = build_graph(); + for (id, node) in &g.nodes { + if let Some(parent) = node.parent { + assert!(g.nodes.contains_key(parent), "node {id} parent {parent}"); + } + } + } + + #[test] + fn graph_terminal_canonical_always_succeeds_for_reddit_paths() { + let cases: &[(&[&str], &str)] = &[ + (&["r", "rust"], "https://reddit.com/r/rust"), + ( + &["r", "rust", "comments", "x"], + "https://reddit.com/r/rust/comments/x", + ), + ]; + for (segs, want) in cases { + let p = test_parts("reddit.com", segs, &[]); + assert_eq!(canon(&p), *want); + } + } + + #[test] + fn breadcrumb_parent_walk_matches_parent_url_semantics() { + let p = test_parts("reddit.com", &["r", "aww", "comments", "id1"], &[]); + let c = crumbs(&p); + assert_eq!(c.len(), 4); + assert_eq!( + c.get(c.len() - 2).map(String::as_str), + Some("https://reddit.com/r/aww") + ); + } + + #[test] + fn youtube_watch_without_v_falls_back_to_generic() { + let p = test_parts("youtube.com", &["watch"], &[]); + assert_eq!(terminal(&p), Some("youtube_watch")); + assert_eq!(canon(&p), "https://youtube.com/watch"); + } + + #[test] + fn reddit_only_comments_path_stops_at_gate() { + let p = test_parts("reddit.com", &["r", "rust", "comments"], &[]); + assert_eq!(terminal(&p), Some("reddit_comments_gate")); + assert_eq!(canon(&p), "https://reddit.com/r/rust"); + } + + #[test] + fn generic_deep_path_many_segments() { + let segs: Vec<&str> = (0..10) + .map(|i| match i { + 0 => "a", + 1 => "b", + 2 => "c", + 3 => "d", + 4 => "e", + 5 => "f", + 6 => "g", + 7 => "h", + 8 => "i", + _ => "j", + }) + .collect(); + let p = test_parts("site.com", &segs, &[]); + assert_eq!(crumbs(&p).len(), 11); + } + + #[test] + fn traverse_literal_r_required_for_subreddit() { + let p = test_parts("reddit.com", &["rust"], &[]); + assert_eq!(terminal(&p), None); + } + + #[test] + fn http_scheme_upgraded_via_generic_fallback_host() { + let parsed = UrlParts::parse("http://example.com/page").unwrap(); + assert_eq!(canon(&parsed), "https://example.com/page"); + } + + #[test] + fn each_graph_node_canonical_is_invokable() { + let g = build_graph(); + let empty = Context::default(); + for (id, node) in &g.nodes { + let _ = (node.canonical)(&empty); + let _ = id; + } + } + + #[test] + fn reddit_double_listing_suffix_both_absorbed() { + let p = test_parts("reddit.com", &["r", "rust", "hot", "new"], &[]); + assert_eq!(terminal(&p), Some("reddit_subreddit")); + assert_eq!(canon(&p), "https://reddit.com/r/rust"); + } + + #[test] + fn youtu_be_empty_path_stays_at_entry() { + let p = test_parts("youtu.be", &[], &[]); + assert_eq!(terminal(&p), Some("youtu_be_entry")); + } +} diff --git a/server/src/url_rules/graph_builder.rs b/server/src/url_rules/graph_builder.rs new file mode 100644 index 0000000000000000000000000000000000000000..243204ad5ca514fff459057956080cacb5bf38e2 --- /dev/null +++ b/server/src/url_rules/graph_builder.rs @@ -0,0 +1,73 @@ +//! Declarative construction of the URL graph with build-time link validation. + +use std::collections::HashMap; + +use super::graph::{CanonicalFn, Edge, EdgePattern, Graph, Node}; + +pub struct GraphBuilder { + nodes: HashMap<&'static str, Node>, + current: Option<&'static str>, +} + +impl GraphBuilder { + pub fn new() -> Self { + Self { + nodes: HashMap::new(), + current: None, + } + } + + pub fn node(mut self, id: &'static str) -> Self { + self.nodes.entry(id).or_insert_with(Node::empty); + self.current = Some(id); + self + } + + pub fn canonical(mut self, f: CanonicalFn) -> Self { + let id = self.current.expect("canonical() without node()"); + self.nodes.get_mut(id).expect("node missing").canonical = f; + self + } + + pub fn parent(mut self, parent_id: &'static str) -> Self { + let id = self.current.expect("parent() without node()"); + self.nodes.get_mut(id).expect("node missing").parent = Some(parent_id); + self + } + + pub fn edge(mut self, pattern: EdgePattern, target: &'static str) -> Self { + let id = self.current.expect("edge() without node()"); + self.nodes + .get_mut(id) + .expect("node missing") + .edges + .push(Edge { pattern, target }); + self + } + + pub fn build(self) -> Graph { + for (id, node) in &self.nodes { + if let Some(parent) = node.parent { + assert!( + self.nodes.contains_key(parent), + "node {id}: parent {parent} does not exist" + ); + } + for edge in &node.edges { + if !matches!( + edge.pattern, + EdgePattern::AbsorbAny | EdgePattern::AbsorbIf(_) + ) { + assert!( + self.nodes.contains_key(edge.target), + "node {id}: edge target {} does not exist", + edge.target + ); + } + } + } + Graph { + nodes: self.nodes, + } + } +} diff --git a/server/src/url_rules/mod.rs b/server/src/url_rules/mod.rs index 9e1445346ce77a49dd6a7e7713bf9c57aef353cc..ba4ac662acc7bd4d50ee34613eb7eb6fccbfb17b 100644 --- a/server/src/url_rules/mod.rs +++ b/server/src/url_rules/mod.rs @@ -1,6 +1,7 @@ //! URL canonicalization and hierarchy via a semantic graph (DFA + generic fallback). mod graph; +mod graph_builder; mod parse; mod registry; diff --git a/server/src/url_rules/parse.rs b/server/src/url_rules/parse.rs new file mode 100644 index 0000000000000000000000000000000000000000..19d0c82feb718817168b8b445fcbfa39cf6aa6ba --- /dev/null +++ b/server/src/url_rules/parse.rs @@ -0,0 +1,169 @@ +//! Parse raw strings into host, path segments, and query (order-independent). + +use std::collections::HashMap; + +use url::Url; + +#[derive(Debug, Clone)] +pub struct UrlParts { + pub scheme: String, + pub host: String, + pub path_segments: Vec, + pub query: HashMap, +} + +impl UrlParts { + pub fn parse(raw: &str) -> Option { + let trimmed = raw.trim(); + if trimmed.is_empty() { + return None; + } + + let with_scheme = if trimmed.contains("://") { + trimmed.to_string() + } else if trimmed.starts_with("r/") || trimmed.starts_with("/r/") { + let rest = trimmed.trim_start_matches('/').trim_start_matches("r/"); + format!("https://reddit.com/r/{rest}") + } else if trimmed.contains('.') && !trimmed.starts_with('/') { + format!("https://{trimmed}") + } else { + trimmed.to_string() + }; + + let url = Url::parse(&with_scheme).ok()?; + let host = url.host_str()?.to_string(); + let path_segments: Vec = url + .path_segments() + .map(|segs| segs.filter(|s| !s.is_empty()).map(str::to_string).collect()) + .unwrap_or_default(); + + let mut query = HashMap::new(); + for (k, v) in url.query_pairs() { + query.insert(k.into_owned(), v.into_owned()); + } + + Some(Self { + scheme: url.scheme().to_string(), + path_segments, + query, + host, + }) + } + + /// Host normalized for graph entry matching (lowercase, aliases). + pub fn match_host(&self) -> String { + normalize_match_host(&self.host) + } +} + +pub fn normalize_match_host(host: &str) -> String { + let h = host + .strip_prefix("www.") + .unwrap_or(host) + .to_ascii_lowercase(); + match h.as_str() { + "old.reddit.com" | "new.reddit.com" => "reddit.com".to_string(), + "m.youtube.com" => "youtube.com".to_string(), + _ => h, + } +} + +pub fn strip_tracking_query(query: &mut HashMap) { + query.retain(|k, _| { + let lower = k.to_ascii_lowercase(); + !(lower.starts_with("utm_") + || matches!( + lower.as_str(), + "fbclid" | "gclid" | "ref" | "ref_src" | "ref_source" | "mc_cid" | "mc_eid" + )) + }); +} + +#[cfg(test)] +pub(crate) fn test_parts(host: &str, segs: &[&str], query: &[(&str, &str)]) -> UrlParts { + UrlParts { + scheme: "https".to_string(), + host: host.to_string(), + path_segments: segs.iter().map(|s| (*s).to_string()).collect(), + query: query + .iter() + .map(|(k, v)| (k.to_string(), v.to_string())) + .collect(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn parse_full_url_splits_host_path_query() { + let p = UrlParts::parse("https://www.youtube.com/watch?v=abc&t=4").unwrap(); + assert_eq!(p.host, "www.youtube.com"); + assert_eq!(p.path_segments, vec!["watch"]); + assert_eq!(p.query.get("v").map(String::as_str), Some("abc")); + assert_eq!(p.query.get("t").map(String::as_str), Some("4")); + } + + #[test] + fn parse_r_shortcut_expands_to_reddit() { + let p = UrlParts::parse("r/rust").unwrap(); + assert_eq!(p.match_host(), "reddit.com"); + assert_eq!(p.path_segments, vec!["r", "rust"]); + } + + #[test] + fn parse_slash_r_shortcut() { + let p = UrlParts::parse("/r/aww").unwrap(); + assert_eq!(p.path_segments, vec!["r", "aww"]); + } + + #[test] + fn parse_schemeless_host_path() { + let p = UrlParts::parse("reddit.com/r/rust/comments/aaa/slug").unwrap(); + assert_eq!(p.match_host(), "reddit.com"); + assert_eq!( + p.path_segments, + vec!["r", "rust", "comments", "aaa", "slug"] + ); + } + + #[test] + fn parse_empty_returns_none() { + assert!(UrlParts::parse("").is_none()); + assert!(UrlParts::parse(" ").is_none()); + } + + #[test] + fn normalize_match_host_reddit_aliases() { + assert_eq!(normalize_match_host("old.reddit.com"), "reddit.com"); + assert_eq!(normalize_match_host("NEW.reddit.com"), "reddit.com"); + assert_eq!(normalize_match_host("www.reddit.com"), "reddit.com"); + } + + #[test] + fn normalize_match_host_youtube_aliases() { + assert_eq!(normalize_match_host("m.youtube.com"), "youtube.com"); + assert_eq!(normalize_match_host("www.youtube.com"), "youtube.com"); + } + + #[test] + fn strip_tracking_query_removes_known_params() { + let mut q = HashMap::from([ + ("v".into(), "1".into()), + ("utm_source".into(), "x".into()), + ("fbclid".into(), "y".into()), + ("ref".into(), "z".into()), + ]); + strip_tracking_query(&mut q); + assert_eq!(q.len(), 1); + assert_eq!(q.get("v").map(String::as_str), Some("1")); + } + + #[test] + fn strip_tracking_query_utm_prefix() { + let mut q = HashMap::from([("utm_campaign".into(), "email".into())]); + strip_tracking_query(&mut q); + assert!(q.is_empty()); + } +} diff --git a/server/src/url_rules/registry_tests.rs b/server/src/url_rules/registry_tests.rs new file mode 100644 index 0000000000000000000000000000000000000000..7b76b550f808f590e011469fdae02adf603b43d2 --- /dev/null +++ b/server/src/url_rules/registry_tests.rs @@ -0,0 +1,195 @@ +//! End-to-end tests for the public registry API (`canonicalize_raw`, breadcrumbs, parent). + +use super::registry::{ + canonicalize_raw, looks_like_url, navigable_breadcrumbs, parent_url, resolve_id, +}; + +fn canon(raw: &str) -> String { + canonicalize_raw(raw).unwrap().canonical +} + +#[test] +fn looks_like_url_positive_cases() { + for raw in [ + "https://reddit.com/r/rust", + "r/rust", + "/r/aww", + "reddit.com/r/x", + "www.example.com/path", + "youtu.be/abc", + ] { + assert!(looks_like_url(raw), "{raw}"); + } +} + +#[test] +fn looks_like_url_negative_cases() { + for raw in ["alpha", "beta", "", "hello world", "no-dots"] { + assert!(!looks_like_url(raw), "{raw}"); + } +} + +#[test] +fn resolve_id_matches_canonicalize_raw() { + let raw = "https://youtu.be/xyz"; + assert_eq!( + resolve_id(raw).as_deref(), + Some(canon(raw).as_str()) + ); +} + +#[test] +fn canonicalize_empty_returns_none() { + assert!(canonicalize_raw("").is_none()); +} + +#[test] +fn alias_when_slug_stripped() { + let r = canonicalize_raw( + "https://reddit.com/r/rust/comments/aaa/very_long_title_slug", + ) + .unwrap(); + assert_eq!(r.canonical, "https://reddit.com/r/rust/comments/aaa"); + assert!(r.alias_of.is_some()); +} + +#[test] +fn alias_none_when_already_canonical() { + let raw = "https://reddit.com/r/rust"; + let r = canonicalize_raw(raw).unwrap(); + assert_eq!(r.canonical, raw); + assert!(r.alias_of.is_none()); +} + +#[test] +fn parent_url_subreddit_under_r_hub() { + assert_eq!( + parent_url("https://reddit.com/r/movies").as_deref(), + Some("https://reddit.com/r") + ); +} + +#[test] +fn parent_url_domain_has_none() { + assert_eq!(parent_url("https://reddit.com").as_deref(), None); +} + +#[test] +fn parent_url_generic_site() { + assert_eq!( + parent_url("https://example.com/a/b").as_deref(), + Some("https://example.com/a") + ); +} + +#[test] +fn breadcrumbs_from_canonical_string_roundtrip() { + let id = "https://reddit.com/r/golang/comments/abc123"; + let crumbs = navigable_breadcrumbs(id); + assert_eq!(crumbs.last().map(String::as_str), Some(id)); +} + +// --- Table: Reddit raw URLs → canonical --- + +#[test] +fn reddit_canonical_matrix() { + let cases: &[(&str, &str)] = &[ + ("r/rust", "https://reddit.com/r/rust"), + ("/r/aww", "https://reddit.com/r/aww"), + ("https://reddit.com/r/rust", "https://reddit.com/r/rust"), + ( + "https://www.reddit.com/r/programming/new", + "https://reddit.com/r/programming", + ), + ( + "https://old.reddit.com/r/test/comments/xyz/slug/", + "https://reddit.com/r/test/comments/xyz", + ), + ( + "reddit.com/r/Movies/comments/abc/Title_Case_Slug", + "https://reddit.com/r/movies/comments/abc", + ), + ]; + for (raw, want) in cases { + assert_eq!(canon(raw), *want, "raw={raw}"); + } +} + +// --- Table: YouTube raw URLs → canonical --- + +#[test] +fn youtube_canonical_matrix() { + let cases: &[(&str, &str)] = &[ + ( + "https://youtube.com/watch?v=abc", + "https://youtube.com/watch?v=abc", + ), + ( + "https://www.youtube.com/watch?v=abc&t=1&feature=share", + "https://youtube.com/watch?v=abc", + ), + ("https://youtu.be/abc", "https://youtube.com/watch?v=abc"), + ( + "https://youtube.com/shorts/abc", + "https://youtube.com/watch?v=abc", + ), + ]; + for (raw, want) in cases { + assert_eq!(canon(raw), *want, "raw={raw}"); + } +} + +// --- Table: generic sites --- + +#[test] +fn generic_canonical_matrix() { + let cases: &[(&str, &str)] = &[ + ( + "https://news.ycombinator.com/item?id=38472", + "https://news.ycombinator.com/item?id=38472", + ), + ( + "https://www.github.com/rust-lang/rust/issues/1?utm_source=x", + "https://github.com/rust-lang/rust/issues/1", + ), + ("https://example.com", "https://example.com"), + ]; + for (raw, want) in cases { + assert_eq!(canon(raw), *want, "raw={raw}"); + } +} + +// --- Phantom /comments/ regression (sorter2-specific) --- + +#[test] +fn phantom_comments_not_in_breadcrumbs_for_post() { + let crumbs = navigable_breadcrumbs("https://reddit.com/r/rust/comments/aaa"); + assert!(!crumbs.iter().any(|c| c.ends_with("/comments"))); +} + +#[test] +fn phantom_comments_not_sibling_of_subreddit_in_breadcrumb_chain() { + let crumbs = navigable_breadcrumbs("https://reddit.com/r/rust/comments/aaa"); + let subs: Vec<_> = crumbs + .iter() + .filter(|c| c.contains("/r/rust") && !c.contains("/comments/")) + .collect(); + assert_eq!(subs, vec!["https://reddit.com/r/rust"]); +} + +// --- Distinct items must stay distinct --- + +#[test] +fn different_posts_different_canonical() { + let a = canon("https://reddit.com/r/rust/comments/aaa"); + let b = canon("https://reddit.com/r/rust/comments/bbb"); + assert_ne!(a, b); +} + +#[test] +fn different_subreddits_different_canonical() { + assert_ne!( + canon("https://reddit.com/r/rust"), + canon("https://reddit.com/r/golang") + ); +}