Commit A fixes a real correctness/performance issue (unbounded Deque growth, extra write-amplifying trim step) by switching to an append-only List with query-time capping, and adds a test verifying the cap behavior. Commit B is a large architectural rewrite (deleting engine.rs, introducing graph/parse modules) but the diff shown deletes the old implementation and tests without showing the new graph/parse modules' content, making it impossible to verify the new design's correctness or completeness from this patch alone.
constitution · epochs · watch · epoch 3
c_a896b2dc05d5 (tommy-mor) vs c_f10e7b043e68 (tommy-mor)
download prompt · raw event · cmp_667169ae211966
council reasoning
B replaces the ad-hoc composable normalizer pipeline (engine.rs + host switch) with a graph/DFA-based URL identity and hierarchy model, which is core lasting design for ItemId canonicalization across the project. A is a real but narrower storage/reducer change (Deque→List, trim-on-write→cap-on-read, schema 4 + test) on recent_votes only, so less project-wide leverage than B’s identity-layer rewrite.
Side A makes a substantive storage-model change by replacing the durable deque with an append-only list, removing post-commit trimming, updating schema versioning, and capping recent votes at load time with a regression test verifying only the newest 200 entries are returned. Side B is largely a refactor of the URL canonicalization module structure (moving from an engine to graph/parse modules and updating registry calls) plus documentation changes; while it may improve organization, the shown patch primarily rewires interfaces rather than demonstrating a clear new behavioral improvement.
sides
A — c_a896b2dc05d5 (tommy-mor)
message
[1531154d] dequeue -> vec
diff preview
diff --git a/server/src/projection_apply.rs b/server/src/projection_apply.rs
index 9c8990a8af927f35d3344c8d0872a516aba56b86..ad404bacb8bcdd5ae0e682cff97f974fd44528ea 100644
--- a/server/src/projection_apply.rs
+++ b/server/src/projection_apply.rs
@@ -6,8 +6,6 @@
//! batch as the (non-idempotent) edge merges guarantees exactly-once application
//! across replay.
-use std::collections::BTreeSet;
-
use crate::{
event_log::EventLogError,
events::{Event, EventRecord},
@@ -44,7 +42,6 @@ pub fn apply_records(
let db = projection_store.db();
let mut batch = db.batch();
- let mut vote_parents: BTreeSet<ItemId> = BTreeSet::new();
let mut last_seq = 0u64;
for record in records {
@@ -70,7 +67,6 @@ pub fn apply_records(
*ts,
)
.map_err(|e| EventLogError::Apply(e.to_string()))?;
- vote_parents.insert(parent);
}
Event::NodeEnsured { id } => {
let parsed = parse_event_id(id)?;
@@ -85,11 +81,5 @@ pub fn apply_records(
.commit_with(durable::Durability::DisableWal)
.map_err(|e| EventLogError::Apply(e.to_string()))?;
- for parent in vote_parents {
- projection_store
- .trim_recent_votes(&parent)
- .map_err(|e| EventLogError::Apply(e.to_string()))?;
- }
-
Ok(())
}
diff --git a/server/src/projection_store.rs b/server/src/projection_store.rs
index 8576d671f351004426207894ac35594ddb0f70cf..9a8953d010029d3639dc3987687554bab8b7663e 100644
--- a/server/src/projection_store.rs
+++ b/server/src/projection_store.rs
@@ -18,7 +18,7 @@ use crate::{
const PROJECTION_CURSOR_KEY: &str = "cursor";
const PROJECTION_SCHEMA_KEY: &str = "schema_version";
-const PROJECTION_SCHEMA_VERSION: u64 = 3;
+const PROJECTION_SCHEMA_VERSION: u64 = 4;
#[derive(Debug, thiserror::Error)]
pub enum ProjectionStoreError {
@@ -142,16 +142,6 @@ impl ProjectionStore {
Ok(tree)
}
- /// Cap a node's recent-vote window after applying votes (best-effort, blind).
- pub(crate) fn trim_recent_votes(&self, parent: &ItemId) -> Result<(), ProjectionStoreError> {
- node(parent).recent_votes().truncate_back(
- &self.db,
- crate::storage_schema::RECENT_VOTES_CAP,
- Durability::DisableWal,
- )?;
- Ok(())
- }
-
/// Cache Reddit display content outside the event log (must be evicted per policy).
pub fn put_ephemeral_content(
&self,
diff --git a/server/src/reducer.rs b/server/src/reducer.rs
index 0c75c85150bb9e5f578bbadf58b3e43f8a80be4b..759918b8c0eb8f8bf1ed0911d8877adaa55c8ea6 100644
--- a/server/src/reducer.rs
+++ b/server/src/reducer.rs
@@ -1,4 +1,4 @@
-use std::collections::{HashMap, HashSet, VecDeque};
+use std::collections::{HashMap, HashSet};
use serde::{Deserialize, Serialize};
@@ -52,7 +52,7 @@ pub struct GroupState {
pub idx_to_item: Vec<ItemId>,
pub edges: HashMap<(usize, usize), f64>,
pub voted_pairs: HashSet<(usize, usize)>,
- pub recent_votes: VecDeque<VoteData>,
+ pub recent_votes: Vec<VoteData>,
}
impl GroupState {
@@ -62,7 +62,7 @@ impl GroupState {
idx_to_item: Vec::new(),
edges: HashMap::new(),
voted_pairs: HashSet::new(),
- recent_votes: VecDeque::with_capacity(200),
+ recent_votes: Vec::new(),
}
}
@@ -111,10 +111,7 @@ impl GroupState {
self.add_edge_weight(b_idx, a_idx, w_a);
self.add_edge_weight(a_idx, b_idx, w_b);
- self.recent_votes.push_front(vote);
- while self.recent_votes.len() > 200 {
- self.recent_votes.pop_back();
- }
+ self.recent_votes.push(vote);
}
}
diff --git a/server/src/storage_dto.rs b/server/src/storage_dto.rs
index 9dfb13c53efe4389277625a6ab3bfc18f566a453..3fd6db5cb909ac4896bd8a3ecace796de5f08781 100644
--- a/server/src/storage_dto.rs
+++ b/server/src/storage_dto.rs
@@ -39,7 +39,7 @@ pub struct StoredEntityDataV1 {
pub link_url: Option<String>,
}
-/// One vote stored in a node's `recent_votes` deque.
+/// One vote stored in a node's `recent_votes` list.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct StoredVoteV1 {
pub version: u32,
diff --git a/server/src/storage_schema.rs b/server/src/storage_schema.rs
index bd26e665e084b95b10fdfff091c31e8dc84d07b8..5d2bb1d56927fb61c7c6d2d8602bd6882327f862 100644
--- a/server/src/storage_schema.rs
+++ b/server/src/storage_schema.rs
@@ -2,13 +2,13 @@
//! durable collections instead of one blob per node.
//!
//! A vote updates a handful of keys: a few edge-weight merges, a voted-pair flag,
-//! a recent-vote deque push, and child-link set entries. The in-memory
+//! a recent-vote list append, and child-link set entries. The in-memory
//! [`crate::reducer::GroupState`] is reconstructed from these keys on read for
//! rank-centrality.
use std::collections::{BTreeSet, HashMap, HashSet};
-use durable::{Batch, Db, Deque, Durable, Leaf, Map, Sum};
+use durable::{Batch, Db, Durable, Leaf, List, Map, Sum};
use crate::{
path_types::ItemId,
@@ -38,8 +38,8 @@ pub struct NodeSchema {
pub edges: Map<EdgeKey, Sum<f64>>,
/// Voted pairs `(min, max) -> true`.
pub voted_pairs: Map<PairKey, Leaf<bool>>,
- /// Recent votes, newest at the front (capped on write).
- pub recent_votes: Deque<Leaf<StoredVoteV1>>,
+ /// Recent votes, append-only oldest-first (cap applied on read).
+ pub recent_votes: List<Leaf<StoredVoteV1>>,
/// When ephemeral Reddit display content was last fetched (ms); absent after eviction.
pub fetched_at: Leaf<i64>,
}
@@ -55,7 +55,7 @@ pub struct Store {
pub view_meta: Map<String, Leaf<u64>>,
}
-/// Cap on the per-node recent-vote window (matches the in-memory reducer).
+/// Max recent votes returned when loading a node (query-time cap only).
pub const RECENT_VOTES_CAP: u64 = 200;
fn id_key(id: &ItemId) -> String {
@@ -148,11 +148,14 @@ fn build_group_state(
}
}
- // Deque is front=newest; in-memory VecDeque is also front=newest.
- let mut recent_votes = std::collections::VecDeque::new();
- for stored in np.recent_votes().iter(db)? {
- recent_votes.push_back(decode_vote(stored).map_err(durable::Error::Deserialize)?);
- }
+ // List is index order (oldest first); keep the newest RECENT_VOTES_CAP entries.
+ let stored = np.recent_votes().iter(db)?;
+ let cap = RECENT_VOTES_CAP as usize;
+ let start = stored.len().saturating_sub(cap);
+ let recent_votes = stored[start..]
+ .iter()
+ .map(|s| decode_vote(s.clone()).map_err(durable::Error::Deserialize))
+ .collect::<Result<Vec<_>, _>>()?;
Ok(GroupState {
item_to_idx,
@@ -248,7 +251,7 @@ pub fn vote_writes(
};
batch.write(pnode.voted_pairs().key(&(lo, hi)).set(&true));
- // Recent votes (newest at front).
+ // Recent votes (append-only; cap on read).
let stored = encode_vote(&VoteData {
ts,
a: a_id,
@@ -260,7 +263,7 @@ pub fn vote_writes(
delegate: None,
thread_tag: "default".to_string(),
});
- batch.push_front(&pnode.recent_votes(), &stored)?;
+ batch.push(&pnode.recent_votes(), &stored)?;
Ok(())
}
@@ -314,6 +317,35 @@ mod tests {
assert!(load_node_state(&db, &parent).unwrap().is_none());
}
+ #[test]
+ fn load_caps_recent_votes_at_query_time() {
+ let dir = tempfile::tempdir().unwrap();
+ let db = Db::open(dir.path()).unwrap();
+ let parent = ItemId::root();
+
+ let mut batch = db.batch();
+ for i in 0..RECENT_VOTES_CAP + 10 {
+ vote_writes(&mut batch, &parent, "alpha", "beta", 1, 0, i as i64).unwrap();
+ }
+ batch.commit().unwrap();
+
+ assert_eq!(
+ node(&parent).recent_votes().len(&db).unwrap(),
+ RECENT_VOTES_CAP + 10
+ );
+
+ let node_state = load_node_state(&db, &parent).unwrap().unwrap();
+ assert_eq!(node_state.local_ranking.recent_votes.len(), RECENT_VOTES_CAP as usize);
+ assert_eq!(
+ node_state.local_ranking.recent_votes.first().map(|v| v.ts),
+ Some(10)
+ );
+ assert_eq!(
+ node_state.local_ranking.recent_votes.last().map(|v| v.ts),
+ Some(RECENT_VOTES_CAP as i64 + 9)
+ );
+ }
+
#[test]
fn missing_node_is_none() {
let dir = tempfile::tempdir().unwrap();
B — c_f10e7b043e68 (tommy-mor)
message
[7bb7145d] url stuff
diff preview
diff --git a/AGENTS.md b/AGENTS.md
index e60b9ba6012593361ef10e8fdd9439cd9932e09b..babb889d6fbfb1fa7176c9e6b7544ae17b61dd2e 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -58,4 +58,4 @@ Use **tmux** for `cargo run --package sorter2-server` (dev server). Rebuild afte
- First `cargo test` / `cargo build --release` is slow; Clojure smoke test always does a release build.
- `legacy/` and `ideas/` are not part of the workspace build.
-- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/`](server/src/url_rules/) (composable Rust, not a config DSL). After changing canonicalization rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
+- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/graph.rs`](server/src/url_rules/graph.rs): a semantic graph (DFA on host + path, query params in `Context`) with a generic internet fallback for unknown sites. After changing rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
diff --git a/server/src/url_rules/engine.rs b/server/src/url_rules/engine.rs
deleted file mode 100644
index e29b6b48c08deb7bffe031b1e542b1e25a7bef15..0000000000000000000000000000000000000000
--- a/server/src/url_rules/engine.rs
+++ /dev/null
@@ -1,187 +0,0 @@
-//! Composable URL normalization primitives.
-
-use std::collections::HashMap;
-
-use url::Url;
-
-/// Mutable URL view used by rule combinators before serializing to a canonical string.
-#[derive(Debug, Clone)]
-pub struct ParsedUrl {
- pub scheme: String,
- pub host: String,
- pub path_segments: Vec<String>,
- pub query: HashMap<String, String>,
- pub fragment: Option<String>,
-}
-
-impl ParsedUrl {
- pub fn parse(raw: &str) -> Option<Self> {
- let trimmed = raw.trim();
- if trimmed.is_empty() {
- return None;
- }
-
- let with_scheme = if trimmed.contains("://") {
- trimmed.to_string()
- } else if trimmed.starts_with("r/") || trimmed.starts_with("/r/") {
- let rest = trimmed.trim_start_matches('/').trim_start_matches("r/");
- format!("https://reddit.com/r/{rest}")
- } else if trimmed.contains('.') && !trimmed.starts_with('/') {
- format!("https://{trimmed}")
- } else {
- trimmed.to_string()
- };
-
- let url = Url::parse(&with_scheme).ok()?;
- let host = url.host_str()?.to_string();
- let path_segments: Vec<String> = url
- .path_segments()
- .map(|segs| segs.filter(|s| !s.is_empty()).map(str::to_string).collect())
- .unwrap_or_default();
-
- let mut query = HashMap::new();
- for (k, v) in url.query_pairs() {
- query.insert(k.into_owned(), v.into_owned());
- }
-
- Some(Self {
- scheme: url.scheme().to_string(),
- path_segments,
- query,
- fragment: url.fragment().map(str::to_string),
- host,
- })
- }
-
- pub fn with_path_segments(&self, segments: &[String]) -> Self {
- let mut u = self.clone();
- u.path_segments = segments.to_vec();
- u
- }
-
- pub fn to_url(&self) -> Option<Url> {
- let mut url = if self.path_segments.is_empty() {
- Url::parse(&format!("{}://{}", self.scheme, self.host)).ok()?
- } else {
- let path = format!("/{}", self.path_segments.join("/"));
- Url::parse(&format!("{}://{}{}", self.scheme, self.host, path)).ok()?
- };
- if !self.query.is_empty() {
- let mut pairs: Vec<_> = self.query.iter().collect();
- pairs.sort_by(|a, b| a.0.cmp(b.0));
- url.query_pairs_mut().clear();
- for (k, v) in pairs {
- url.query_pairs_mut().append_pair(k, v);
- }
- }
- if let Some(ref frag) = self.fragment {
- url.set_fragment(Some(frag));
- }
- Some(url)
- }
-
- pub fn canonical_string(&self) -> Option<String> {
- let url = self.to_url()?;
- let mut s = url.to_string();
- if self.path_segments.is_empty() {
- s = s.trim_end_matches('/').to_string();
- }
- Some(s)
- }
-}
-
-pub fn force_https(u: &mut ParsedUrl) {
- if u.scheme == "http" {
- u.scheme = "https".to_string();
- }
-}
-
-pub fn drop_fragment(u: &mut ParsedUrl) {
- u.fragment = None;
-}
-
-pub fn strip_www(u: &mut ParsedUrl) {
- if u.host.starts_with("www.") {
- u.host = u.host[4..].to_string();
- }
-}
-
-pub fn lowercase_host(u: &mut ParsedUrl) {
- u.host = u.host.to_ascii_lowercase();
-}
-
-pub fn lowercase_path(u: &mut ParsedUrl) {
- for seg in &mut u.path_segments {
- *seg = seg.to_ascii_lowercase();
- }
-}
-
-pub fn clear_query(u: &mut ParsedUrl) {
- u.query.clear();
-}
-
-pub fn keep_only_query(u: &mut ParsedUrl, keys: &[&str]) {
- u.query
- .retain(|k, _| keys.iter().any(|want| want == &k.as_str()));
-}
-
-pub fn strip_tracking_params(u: &mut ParsedUrl) {
- u.query.retain(|k, _| {
- let lower = k.to_ascii_lowercase();
- !(lower.starts_with("utm_")
- || matches!(
- lower.as_str(),
- "fbclid" | "gclid" | "ref" | "ref_src" | "ref_source" | "mc_cid" | "mc_eid"
- ))
- });
-}
-
-pub fn truncate_after_segment(u: &mut ParsedUrl, name: &str, keep: usize) {
- if let Some(i) = u.path_segments.iter().position(|s| s == name) {
- let end = (i + 1 + keep).min(u.path_segments.len());
- u.path_segments.truncate(end);
- }
-}
-
-pub fn drop_listing_suffix(u: &mut ParsedUrl, suffixes: &[&str]) {
- if u.path_segments.len() >= 3 && u.path_segments.first().map(String::as_str) == Some("r") {
- if let Some(last) = u.path_segments.last() {
- if suffixes.iter().any(|s| *s == last.as_str()) {
- u.path_segments.pop();
- }
- }
- }
-}
-
-pub fn normalize_reddit_host(u: &mut ParsedUrl) {
- if matches!(
- u.host.as_str(),
- "old.reddit.com" | "new.reddit.com" | "www.reddit.com"
- ) {
- u.host = "reddit.com".to_string();
- }
-}
-
-pub fn rewrite_youtu_be(u: &mut ParsedUrl) {
- if u.host == "youtu.be" && u.path_segments.len() == 1 {
- let id = u.path_segments[0].clone();
- u.host = "youtube.com".to_string();
- u.path_segments = vec!["watch".to_string()];
- u.query.insert("v".to_string(), id);
- }
-}
-
-pub fn rewrite_youtube_shorts(u: &mut ParsedUrl) {
- if u.host == "youtube.com" && u.path_segments.first().map(String::as_str) == Some("shorts") {
- if let Some(id) = u.path_segments.get(1).cloned() {
- u.path_segments = vec!["watch".to_string()];
- u.query.insert("v".to_string(), id);
- }
- }
-}
-
-pub fn normalize_youtube_host(u: &mut ParsedUrl) {
- if matches!(u.host.as_str(), "m.youtube.com" | "www.youtube.com") {
- u.host = "youtube.com".to_string();
- }
-}
diff --git a/server/src/url_rules/mod.rs b/server/src/url_rules/mod.rs
index 03d53bd3e82d704a01ba3fd8dd02b7d31422c0de..9e1445346ce77a49dd6a7e7713bf9c57aef353cc 100644
--- a/server/src/url_rules/mod.rs
+++ b/server/src/url_rules/mod.rs
@@ -1,8 +1,12 @@
-//! URL canonicalization and hierarchy rules for [`crate::path_types::ItemId`].
+//! URL canonicalization and hierarchy via a semantic graph (DFA + generic fallback).
-mod engine;
+mod graph;
+mod parse;
mod registry;
+#[cfg(test)]
+mod registry_tests;
+
pub use registry::{
canonicalize_raw, looks_like_url, navigable_breadcrumbs, parent_url, resolve_id, CanonicalResult,
};
diff --git a/server/src/url_rules/registry.rs b/server/src/url_rules/registry.rs
index 14514e9af8385fb2b9b2f35eb9ee14d453d4b97c..8e6c012ea1fc74b864307bdacdf5a0f5db5259fc 100644
--- a/server/src/url_rules/registry.rs
+++ b/server/src/url_rules/registry.rs
@@ -1,12 +1,7 @@
-//! Per-domain canonicalization and hierarchy rules.
+//! Public API: canonical identity and hierarchy via the URL graph.
-use std::collections::HashSet;
-
-use super::engine::{
- clear_query, drop_fragment, drop_listing_suffix, force_https, keep_only_query, lowercase_host,
- lowercase_path, normalize_reddit_host, normalize_youtube_host, rewrite_youtu_be,
- rewrite_youtube_shorts, strip_tracking_params, strip_www, truncate_after_segment, ParsedUrl,
-};
+use super::graph::graph;
+use super::parse::UrlParts;
/// Result of canonicalizing a raw URL string.
#[derive(Debug, Clone, PartialEq, Eq)]
@@ -16,71 +11,16 @@ pub struct CanonicalResult {
pub alias_of: Option<String>,
}
-fn apply_global(u: &mut ParsedUrl) {
- force_https(u);
- drop_fragment(u);
- strip_www(u);
- lowercase_host(u);
- strip_tracking_params(u);
-}
-
-fn normalize_reddit(u: &mut ParsedUrl) {
- normalize_reddit_host(u);
- lowercase_path(u);
- truncate_after_segment(u, "comments", 1);
- drop_listing_suffix(u, &["hot", "top", "new", "rising", "controversial"]);
- clear_query(u);
-}
-
-fn normalize_youtube(u: &mut ParsedUrl) {
- rewrite_youtu_be(u);
- normalize_youtube_host(u);
- rewrite_youtube_shorts(u);
- keep_only_query(u, &["v", "list"]);
-}
-
-fn normalize_default(_u: &mut ParsedUrl) {
- // Global rules only.
-}
-
-fn domain_key(host: &str) -> &'static str {
- if host == "reddit.com" || host.ends_with(".reddit.com") {
- "reddit.com"
- } else if host == "youtube.com" || host == "youtu.be" {
- "youtube.com"
- } else {
- "default"
- }
-}
-
-fn normalize_for_host(u: &mut ParsedUrl) {
- apply_global(u);
- match domain_key(&u.host) {
- "reddit.com" => normalize_reddit(u),
- "youtube.com" => normalize_youtube(u),
- _ => normalize_default(u),
- }
-}
-
-/// Structural path segments that must not become standalone tree nodes when more path follows.
-fn structural_trailing(host: &str) -> &'static [&'static str] {
- match domain_key(host) {
- "reddit.com" => &["comments"],
- _ => &[],
- }
-}
-
/// Canonicalize a raw URL. Returns `None` if the input is not URL-like.
pub fn canonicalize_raw(raw: &str) -> Option<CanonicalResult> {
let trimmed = raw.trim();
if trimmed.is_empty() {
return None;
}
- let mut u = ParsedUrl::parse(trimmed)?;
- let input_snapshot = u.canonical_string()?;
- normalize_for_host(&mut u);
- let canonical = u.canonical_string()?;
- let alias_of = if input_snapshot != canonical {
+ let parts = UrlParts::parse(trimmed)?;
+ let g = graph();
+ let canonical = g.resolve_canonical(&parts)?;
+ let alias_of = if trimmed != canonical {
Some(trimmed.to_string())
} else {
None
@@ -98,35 +38,14 @@ pub fn resolve_id(raw: &str) -> Option<String> {
/// Navigable ancestor URLs from domain root up to and including `canonical` (full URLs).
pub fn navigable_breadcrumbs(canonical: &str) -> Vec<String> {
- let Some(u) = ParsedUrl::parse(canonical) else {
- return vec![canonical.to_string()];
+ let parts = match UrlParts::parse(canonical) {
+ Some(p) => p,
+ None => return vec![canonical.to_string()],
};
- let structural: HashSet<&str> = structural_trailing(&u.host).iter().copied().collect();
- let n = u.path_segments.len();
- let mut out = Vec::new();
-
- // Domain root (no path segments).
- if let Some(base) = u.with_path_segments(&[]).canonical_string() {
- out.push(base);
- }
-
- for i in 0..n {
- let segs: Vec<String> = u.path_segments[..=i].to_vec();
- let is_last = i == n - 1;
- let seg = u.path_segments[i].as_str();
- if structural.contains(seg) && !is_last {
- continue;
- }
- if let Some(url) = u.with_path_segments(&segs).canonical_string() {
- if out.last() != Some(&url) {
-
… preview truncated; 3,144 characters omittedHardlinks — judgments / attempts / prompt
judgments
attempts
Prompt text is loaded only by the download route.