Side A is a focused, well-documented fix that removes duplicate UI logic (ExpandNewThreadForm) and unifies the home page's compose slot with the room page pattern, including removal of the now-obsolete test — a clear, verifiable improvement in code clarity and correctness. Side B's diff deletes the old URL engine and tests but references new modules (graph.rs, parse.rs, registry_tests.rs) that are not shown in the patch, making the actual substance of the replacement unverifiable and the vague commit message ('url stuff') further undermines confidence in its completeness and intent.
constitution · epochs · watch · epoch 3
c_f515f8a12d7a (tommy-mor) vs c_f10e7b043e68 (tommy-mor)
download prompt · raw event · cmp_5a43630e4de28c
council reasoning
B replaces the ad-hoc ParsedUrl combinators and per-host normalize paths with a lasting graph/DFA-based canonicalization design (and thins registry to that API), which underpins ItemId identity project-wide. A is a correct, focused UX cleanup (SSR #new-thread-ui-slot on home and delete ExpandNewThreadForm + toolbar), but it only simplifies one compose flow rather than core URL architecture.
Side B replaces the URL canonicalization implementation with a new graph- and parser-based architecture, rewires the public registry API to use it, and reorganizes the module structure, representing a foundational redesign of a core subsystem. Side A is a targeted UI simplification that removes the `ExpandNewThreadForm` action and server-renders the collapsed compose/login hint on the home page, which reduces duplication but has a narrower, feature-specific impact.
sides
A — c_f515f8a12d7a (tommy-mor)
message
[601d3a05] fix(html): drop home toolbar + and ExpandNewThreadForm (single + flow) Public home now SSRs #new-thread-ui-slot like room pages: collapsed compose for signed-in users, login hint when logged out. Removes the extra toolbar that morphed the same collapsed state and the expand_new_thread_form action. Made-with: Cursor
diff preview
diff --git a/server/src/api/ui_html.rs b/server/src/api/ui_html.rs
index e979053ff1ba8c0e2bf55add0a32b7de11cf1e56..f3ce5cb2ab2f923440a8479d0f1fb4acbba166ca 100644
--- a/server/src/api/ui_html.rs
+++ b/server/src/api/ui_html.rs
@@ -139,51 +139,6 @@ async fn dispatch_ui_action(
}
}
}
- HtmlUiAction::ExpandNewThreadForm { room_wire } => {
- let room_wire = room_wire.trim().to_string();
- if room_wire.is_empty() {
- return ui_js_warn("missing room").into_response();
- }
- if room_wire == "public" {
- let reduced = state.reduced.read().await;
- let user = session.map(|s| s.username.as_str());
- drop(reduced);
- let markup = if user.is_some() {
- fragment_new_thread_slot(&ThreadNav::public(), true, false)
- } else {
- login_to_post_hint_markup()
- };
- return JsBuilder::new()
- .morph_inner_selector("#new-thread-ui-slot", markup)
- .into_response();
- }
- let reduced = state.reduced.read().await;
- let user = session.map(|s| s.username.as_str());
- if !reduced.rooms.contains(&room_wire) {
- drop(reduced);
- return ui_js_warn("room not found").into_response();
- }
- if !user_can_view_room(&reduced, &room_wire, user) {
- drop(reduced);
- return ui_js_warn("forbidden").into_response();
- }
- let can_post = session
- .as_ref()
- .map(|s| user_can_post_room(&reduced, &room_wire, &s.username))
- .unwrap_or(false);
- drop(reduced);
- let Some(nav) = ThreadNav::from_room_id(&room_wire) else {
- return ui_js_warn("bad room").into_response();
- };
- let markup = if can_post {
- fragment_new_thread_slot(&nav, true, false)
- } else {
- login_to_post_hint_markup()
- };
- JsBuilder::new()
- .morph_inner_selector("#new-thread-ui-slot", markup)
- .into_response()
- }
HtmlUiAction::SetRoomMembersExpanded { room_wire, expanded } => {
let room_wire = room_wire.trim().to_string();
if room_wire.is_empty() {
diff --git a/server/src/html/forum/feed.rs b/server/src/html/forum/feed.rs
index 1b4ae7baa3ad4757b74172d7b67f3f2b33d1075d..945bdd6c48bc164e2cf91fd0996c4321b75f5abf 100644
--- a/server/src/html/forum/feed.rs
+++ b/server/src/html/forum/feed.rs
@@ -14,6 +14,7 @@ use crate::timeago;
use super::ingest::ingest_entry_markup;
use super::nav::ThreadNav;
+use super::new_thread::{fragment_new_thread_slot, login_to_post_hint_markup};
use super::page::auth_strip;
use super::paginator::{render_thread_paginator, PAGE_SIZE};
use crate::html::{
@@ -217,9 +218,6 @@ pub async fn home(
let strip = auth_strip(&headers, &jar, &reduced_read);
drop(reduced_read);
- use crate::html::ui_action::{HtmlUiAction, UI_RPC_FIELD};
- use crate::form_template::template_json_compact;
-
let page = layout(
"slug.social",
"view-thread",
@@ -243,15 +241,13 @@ pub async fn home(
}
}
p class="muted" { "dark = time-ordered · light = vote-ranked" }
- div class="thread-feed-toolbar" {
- form method="POST" action="/ui" {
- input type="hidden" name=(UI_RPC_FIELD) value=(template_json_compact(&HtmlUiAction::ExpandNewThreadForm {
- room_wire: "public".into(),
- }).expect("static json"));
- button type="submit" class="section-add-btn" { "+" }
+ div id="new-thread-ui-slot" {
+ @if user.is_some() {
+ (fragment_new_thread_slot(&nav, true, false))
+ } @else {
+ (login_to_post_hint_markup())
}
}
- div id="new-thread-ui-slot" {}
(render_thread_feed(Some(&nav), "thread-feed", &public_rows, now))
(cli_panel(&["npx slugsocial public forum list"]))
},
diff --git a/server/src/html/ui_action.rs b/server/src/html/ui_action.rs
index 5031ebfeb928f28e471f23210b8c644654adb7c7..da0c9b3541e8e4988a78768cddef22324755c3f2 100644
--- a/server/src/html/ui_action.rs
+++ b/server/src/html/ui_action.rs
@@ -38,11 +38,6 @@ pub enum HtmlUiAction {
RedactPost {
post_id: String,
},
- /// Morph `#new-thread-ui-slot` inner to the collapsed compose toggle (or login hint).
- /// Use `room_wire: "public"` for the public forum home; otherwise a private room id (`short/slug`).
- ExpandNewThreadForm {
- room_wire: String,
- },
/// Morph `#room-members-section` — members list open or collapsed (server-rendered).
SetRoomMembersExpanded {
room_wire: String,
@@ -131,26 +126,6 @@ mod tests {
);
}
- #[test]
- fn expand_new_thread_form_public() {
- let template = serde_json::json!({
- "action": "expand_new_thread_form",
- "room_wire": "public",
- });
- let mut form = HashMap::new();
- form.insert(
- UI_RPC_FIELD.to_string(),
- serde_json::to_string(&template).unwrap(),
- );
- let a = parse_html_ui_from_form(&form).unwrap();
- assert_eq!(
- a,
- HtmlUiAction::ExpandNewThreadForm {
- room_wire: "public".into(),
- }
- );
- }
-
#[test]
fn expand_post_full_round_trip() {
let template = serde_json::json!({
B — c_f10e7b043e68 (tommy-mor)
message
[7bb7145d] url stuff
diff preview
diff --git a/AGENTS.md b/AGENTS.md
index e60b9ba6012593361ef10e8fdd9439cd9932e09b..babb889d6fbfb1fa7176c9e6b7544ae17b61dd2e 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -58,4 +58,4 @@ Use **tmux** for `cargo run --package sorter2-server` (dev server). Rebuild afte
- First `cargo test` / `cargo build --release` is slow; Clojure smoke test always does a release build.
- `legacy/` and `ideas/` are not part of the workspace build.
-- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/`](server/src/url_rules/) (composable Rust, not a config DSL). After changing canonicalization rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
+- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/graph.rs`](server/src/url_rules/graph.rs): a semantic graph (DFA on host + path, query params in `Context`) with a generic internet fallback for unknown sites. After changing rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
diff --git a/server/src/url_rules/engine.rs b/server/src/url_rules/engine.rs
deleted file mode 100644
index e29b6b48c08deb7bffe031b1e542b1e25a7bef15..0000000000000000000000000000000000000000
--- a/server/src/url_rules/engine.rs
+++ /dev/null
@@ -1,187 +0,0 @@
-//! Composable URL normalization primitives.
-
-use std::collections::HashMap;
-
-use url::Url;
-
-/// Mutable URL view used by rule combinators before serializing to a canonical string.
-#[derive(Debug, Clone)]
-pub struct ParsedUrl {
- pub scheme: String,
- pub host: String,
- pub path_segments: Vec<String>,
- pub query: HashMap<String, String>,
- pub fragment: Option<String>,
-}
-
-impl ParsedUrl {
- pub fn parse(raw: &str) -> Option<Self> {
- let trimmed = raw.trim();
- if trimmed.is_empty() {
- return None;
- }
-
- let with_scheme = if trimmed.contains("://") {
- trimmed.to_string()
- } else if trimmed.starts_with("r/") || trimmed.starts_with("/r/") {
- let rest = trimmed.trim_start_matches('/').trim_start_matches("r/");
- format!("https://reddit.com/r/{rest}")
- } else if trimmed.contains('.') && !trimmed.starts_with('/') {
- format!("https://{trimmed}")
- } else {
- trimmed.to_string()
- };
-
- let url = Url::parse(&with_scheme).ok()?;
- let host = url.host_str()?.to_string();
- let path_segments: Vec<String> = url
- .path_segments()
- .map(|segs| segs.filter(|s| !s.is_empty()).map(str::to_string).collect())
- .unwrap_or_default();
-
- let mut query = HashMap::new();
- for (k, v) in url.query_pairs() {
- query.insert(k.into_owned(), v.into_owned());
- }
-
- Some(Self {
- scheme: url.scheme().to_string(),
- path_segments,
- query,
- fragment: url.fragment().map(str::to_string),
- host,
- })
- }
-
- pub fn with_path_segments(&self, segments: &[String]) -> Self {
- let mut u = self.clone();
- u.path_segments = segments.to_vec();
- u
- }
-
- pub fn to_url(&self) -> Option<Url> {
- let mut url = if self.path_segments.is_empty() {
- Url::parse(&format!("{}://{}", self.scheme, self.host)).ok()?
- } else {
- let path = format!("/{}", self.path_segments.join("/"));
- Url::parse(&format!("{}://{}{}", self.scheme, self.host, path)).ok()?
- };
- if !self.query.is_empty() {
- let mut pairs: Vec<_> = self.query.iter().collect();
- pairs.sort_by(|a, b| a.0.cmp(b.0));
- url.query_pairs_mut().clear();
- for (k, v) in pairs {
- url.query_pairs_mut().append_pair(k, v);
- }
- }
- if let Some(ref frag) = self.fragment {
- url.set_fragment(Some(frag));
- }
- Some(url)
- }
-
- pub fn canonical_string(&self) -> Option<String> {
- let url = self.to_url()?;
- let mut s = url.to_string();
- if self.path_segments.is_empty() {
- s = s.trim_end_matches('/').to_string();
- }
- Some(s)
- }
-}
-
-pub fn force_https(u: &mut ParsedUrl) {
- if u.scheme == "http" {
- u.scheme = "https".to_string();
- }
-}
-
-pub fn drop_fragment(u: &mut ParsedUrl) {
- u.fragment = None;
-}
-
-pub fn strip_www(u: &mut ParsedUrl) {
- if u.host.starts_with("www.") {
- u.host = u.host[4..].to_string();
- }
-}
-
-pub fn lowercase_host(u: &mut ParsedUrl) {
- u.host = u.host.to_ascii_lowercase();
-}
-
-pub fn lowercase_path(u: &mut ParsedUrl) {
- for seg in &mut u.path_segments {
- *seg = seg.to_ascii_lowercase();
- }
-}
-
-pub fn clear_query(u: &mut ParsedUrl) {
- u.query.clear();
-}
-
-pub fn keep_only_query(u: &mut ParsedUrl, keys: &[&str]) {
- u.query
- .retain(|k, _| keys.iter().any(|want| want == &k.as_str()));
-}
-
-pub fn strip_tracking_params(u: &mut ParsedUrl) {
- u.query.retain(|k, _| {
- let lower = k.to_ascii_lowercase();
- !(lower.starts_with("utm_")
- || matches!(
- lower.as_str(),
- "fbclid" | "gclid" | "ref" | "ref_src" | "ref_source" | "mc_cid" | "mc_eid"
- ))
- });
-}
-
-pub fn truncate_after_segment(u: &mut ParsedUrl, name: &str, keep: usize) {
- if let Some(i) = u.path_segments.iter().position(|s| s == name) {
- let end = (i + 1 + keep).min(u.path_segments.len());
- u.path_segments.truncate(end);
- }
-}
-
-pub fn drop_listing_suffix(u: &mut ParsedUrl, suffixes: &[&str]) {
- if u.path_segments.len() >= 3 && u.path_segments.first().map(String::as_str) == Some("r") {
- if let Some(last) = u.path_segments.last() {
- if suffixes.iter().any(|s| *s == last.as_str()) {
- u.path_segments.pop();
- }
- }
- }
-}
-
-pub fn normalize_reddit_host(u: &mut ParsedUrl) {
- if matches!(
- u.host.as_str(),
- "old.reddit.com" | "new.reddit.com" | "www.reddit.com"
- ) {
- u.host = "reddit.com".to_string();
- }
-}
-
-pub fn rewrite_youtu_be(u: &mut ParsedUrl) {
- if u.host == "youtu.be" && u.path_segments.len() == 1 {
- let id = u.path_segments[0].clone();
- u.host = "youtube.com".to_string();
- u.path_segments = vec!["watch".to_string()];
- u.query.insert("v".to_string(), id);
- }
-}
-
-pub fn rewrite_youtube_shorts(u: &mut ParsedUrl) {
- if u.host == "youtube.com" && u.path_segments.first().map(String::as_str) == Some("shorts") {
- if let Some(id) = u.path_segments.get(1).cloned() {
- u.path_segments = vec!["watch".to_string()];
- u.query.insert("v".to_string(), id);
- }
- }
-}
-
-pub fn normalize_youtube_host(u: &mut ParsedUrl) {
- if matches!(u.host.as_str(), "m.youtube.com" | "www.youtube.com") {
- u.host = "youtube.com".to_string();
- }
-}
diff --git a/server/src/url_rules/mod.rs b/server/src/url_rules/mod.rs
index 03d53bd3e82d704a01ba3fd8dd02b7d31422c0de..9e1445346ce77a49dd6a7e7713bf9c57aef353cc 100644
--- a/server/src/url_rules/mod.rs
+++ b/server/src/url_rules/mod.rs
@@ -1,8 +1,12 @@
-//! URL canonicalization and hierarchy rules for [`crate::path_types::ItemId`].
+//! URL canonicalization and hierarchy via a semantic graph (DFA + generic fallback).
-mod engine;
+mod graph;
+mod parse;
mod registry;
+#[cfg(test)]
+mod registry_tests;
+
pub use registry::{
canonicalize_raw, looks_like_url, navigable_breadcrumbs, parent_url, resolve_id, CanonicalResult,
};
diff --git a/server/src/url_rules/registry.rs b/server/src/url_rules/registry.rs
index 14514e9af8385fb2b9b2f35eb9ee14d453d4b97c..8e6c012ea1fc74b864307bdacdf5a0f5db5259fc 100644
--- a/server/src/url_rules/registry.rs
+++ b/server/src/url_rules/registry.rs
@@ -1,12 +1,7 @@
-//! Per-domain canonicalization and hierarchy rules.
+//! Public API: canonical identity and hierarchy via the URL graph.
-use std::collections::HashSet;
-
-use super::engine::{
- clear_query, drop_fragment, drop_listing_suffix, force_https, keep_only_query, lowercase_host,
- lowercase_path, normalize_reddit_host, normalize_youtube_host, rewrite_youtu_be,
- rewrite_youtube_shorts, strip_tracking_params, strip_www, truncate_after_segment, ParsedUrl,
-};
+use super::graph::graph;
+use super::parse::UrlParts;
/// Result of canonicalizing a raw URL string.
#[derive(Debug, Clone, PartialEq, Eq)]
@@ -16,71 +11,16 @@ pub struct CanonicalResult {
pub alias_of: Option<String>,
}
-fn apply_global(u: &mut ParsedUrl) {
- force_https(u);
- drop_fragment(u);
- strip_www(u);
- lowercase_host(u);
- strip_tracking_params(u);
-}
-
-fn normalize_reddit(u: &mut ParsedUrl) {
- normalize_reddit_host(u);
- lowercase_path(u);
- truncate_after_segment(u, "comments", 1);
- drop_listing_suffix(u, &["hot", "top", "new", "rising", "controversial"]);
- clear_query(u);
-}
-
-fn normalize_youtube(u: &mut ParsedUrl) {
- rewrite_youtu_be(u);
- normalize_youtube_host(u);
- rewrite_youtube_shorts(u);
- keep_only_query(u, &["v", "list"]);
-}
-
-fn normalize_default(_u: &mut ParsedUrl) {
- // Global rules only.
-}
-
-fn domain_key(host: &str) -> &'static str {
- if host == "reddit.com" || host.ends_with(".reddit.com") {
- "reddit.com"
- } else if host == "youtube.com" || host == "youtu.be" {
- "youtube.com"
- } else {
- "default"
- }
-}
-
-fn normalize_for_host(u: &mut ParsedUrl) {
- apply_global(u);
- match domain_key(&u.host) {
- "reddit.com" => normalize_reddit(u),
- "youtube.com" => normalize_youtube(u),
- _ => normalize_default(u),
- }
-}
-
-/// Structural path segments that must not become standalone tree nodes when more path follows.
-fn structural_trailing(host: &str) -> &'static [&'static str] {
- match domain_key(host) {
- "reddit.com" => &["comments"],
- _ => &[],
- }
-}
-
/// Canonicalize a raw URL. Returns `None` if the input is not URL-like.
pub fn canonicalize_raw(raw: &str) -> Option<CanonicalResult> {
let trimmed = raw.trim();
if trimmed.is_empty() {
return None;
}
- let mut u = ParsedUrl::parse(trimmed)?;
- let input_snapshot = u.canonical_string()?;
- normalize_for_host(&mut u);
- let canonical = u.canonical_string()?;
- let alias_of = if input_snapshot != canonical {
+ let parts = UrlParts::parse(trimmed)?;
+ let g = graph();
+ let canonical = g.resolve_canonical(&parts)?;
+ let alias_of = if trimmed != canonical {
Some(trimmed.to_string())
} else {
None
@@ -98,35 +38,14 @@ pub fn resolve_id(raw: &str) -> Option<String> {
/// Navigable ancestor URLs from domain root up to and including `canonical` (full URLs).
pub fn navigable_breadcrumbs(canonical: &str) -> Vec<String> {
- let Some(u) = ParsedUrl::parse(canonical) else {
- return vec![canonical.to_string()];
+ let parts = match UrlParts::parse(canonical) {
+ Some(p) => p,
+ None => return vec![canonical.to_string()],
};
- let structural: HashSet<&str> = structural_trailing(&u.host).iter().copied().collect();
- let n = u.path_segments.len();
- let mut out = Vec::new();
-
- // Domain root (no path segments).
- if let Some(base) = u.with_path_segments(&[]).canonical_string() {
- out.push(base);
- }
-
- for i in 0..n {
- let segs: Vec<String> = u.path_segments[..=i].to_vec();
- let is_last = i == n - 1;
- let seg = u.path_segments[i].as_str();
- if structural.contains(seg) && !is_last {
- continue;
- }
- if let Some(url) = u.with_path_segments(&segs).canonical_string() {
- if out.last() != Some(&url) {
-
… preview truncated; 3,144 characters omittedHardlinks — judgments / attempts / prompt
judgments
attempts
Prompt text is loaded only by the download route.