Side A is a self-contained, verifiable improvement: it consolidates multiple copy buttons into one panel, adds a real safety invariant (assert_cli_panel_cmd_js_single_quote_safe) preventing broken/unsafe inline JS, and consistently updates all call sites plus both theme CSS files. Side B deletes a working combinator-based engine.rs and its test suite but references new `graph`/`parse` modules that aren't shown being created in this diff, making the change impossible to verify for correctness or completeness, and the terse 'url stuff' message signals low care despite touching core canonicalization logic.
constitution · epochs · watch · epoch 3
c_ca72f0995396 (tommy-mor) vs c_f10e7b043e68 (tommy-mor)
download prompt · raw event · cmp_8cc2a0aabf4c23
council reasoning
B replaces ad-hoc per-domain combinators (engine.rs + large registry normalize path) with a graph/DFA-based URL canonicalization and hierarchy model, which is core ItemId identity infrastructure. A is a solid but peripheral UX polish: multi-command cli_panel grouping, click-to-copy rows, CSS, and JS string safety asserts—real value, but not foundational like B’s redesign.
Side A makes a concrete, user-visible improvement by redesigning `cli_panel` to support grouped commands, click-to-copy rows, and adding assertions that prevent unsafe characters from being embedded in the generated single-quoted JavaScript, while updating all call sites and CSS accordingly. Side B mostly reorganizes the URL canonicalization subsystem by replacing the old engine with new modules and updating registry calls, but the shown patch is largely a refactor/deletion without enough visible new behavior to outweigh A's clear functional and defensive improvements.
sides
A — c_ca72f0995396 (tommy-mor)
message
[798c764d] feat(html): grouped cli_panel with hover-to-copy and JS-safe asserts Single bordered panel for multiple commands; rows copy on click without a separate copy control. Assert CLI strings contain no chars that would break single-quoted onclick JS. Made-with: Cursor
diff preview
diff --git a/server/src/html/forum.rs b/server/src/html/forum.rs
index f6e45b05bb92126966e304619f80ef1d45be7a4b..075860914a6ce18bb0dfa73dc5651ef1a6318b67 100644
--- a/server/src/html/forum.rs
+++ b/server/src/html/forum.rs
@@ -718,7 +718,7 @@ pub async fn home(
}
div id="public-new-thread-ui-slot" {}
(render_thread_feed(Some(&nav), "thread-feed", &public_rows, now))
- (cli_panel("npx slugsocial public forum list"))
+ (cli_panel(&["npx slugsocial public forum list"]))
},
None,
theme_from_jar(&jar),
@@ -868,7 +868,7 @@ async fn thread_view_inner(
div id="thread-live-region" {
(compose_form(&nav, &tag, show_compose))
}
- (cli_panel(&cli))
+ (cli_panel(std::slice::from_ref(&cli)))
},
None,
theme_from_jar(&jar),
@@ -984,9 +984,7 @@ pub async fn room_page(
(new_thread_form_for_room(&nav, true, false))
}
}
- (cli_panel(&forum_cli))
- (cli_panel(&garden_cli))
- (cli_panel(&audit_cli))
+ (cli_panel(&[forum_cli, garden_cli, audit_cli]))
},
None,
theme_from_jar(&jar),
@@ -1409,7 +1407,7 @@ pub async fn user_profile_page(
}
}
}
- (cli_panel(&format!("npx slugsocial public forum list")))
+ (cli_panel(&[format!("npx slugsocial public forum list")]))
},
None,
theme_from_jar(&jar),
diff --git a/server/src/html/garden.rs b/server/src/html/garden.rs
index 9b4789a79c3e293cbfc5f033a0eac8650320d94d..c8ce7de450f7d14e04020511e7eb7323bd487deb 100644
--- a/server/src/html/garden.rs
+++ b/server/src/html/garden.rs
@@ -139,7 +139,7 @@ pub async fn garden_index(
}
}
}
- (cli_panel("npx slugsocial garden tree"))
+ (cli_panel(&["npx slugsocial garden tree"]))
},
None,
theme_from_jar(&jar),
@@ -542,7 +542,7 @@ async fn render_scope_view(
ScopeId::Public => format!("npx slugsocial public garden body {}", path.as_str().trim_start_matches("https://slug.social/~/")),
ScopeId::Room(room_id) => format!("npx slugsocial private {room_id} garden body {}", path.as_str().trim_start_matches("https://slug.social/~/")),
};
- (cli_panel(&cli))
+ (cli_panel(std::slice::from_ref(&cli)))
},
None,
theme_from_jar(&jar),
diff --git a/server/src/html/mod.rs b/server/src/html/mod.rs
index e7f5bfb7a4b2dee2adf96447224c58d641092124..6617781a2e8e86c2e2693788ea7cd0eb0e3659a2 100644
--- a/server/src/html/mod.rs
+++ b/server/src/html/mod.rs
@@ -599,18 +599,38 @@ pub(super) fn render_linkified_with_embeds_in_scope(raw: &str, garden_prefix: &s
}
}
-/// Small CLI hint panel showing how to look up this page from the terminal.
-pub(super) fn cli_panel(cmd: &str) -> Markup {
+/// CLI strings are embedded in a single-quoted JS literal; they must never need escaping.
+fn assert_cli_panel_cmd_js_single_quote_safe(s: &str) {
+ assert!(
+ !s.contains('\\')
+ && !s.contains('\'')
+ && !s.contains('\n')
+ && !s.contains('\r'),
+ "cli_panel cmd must not contain `\\`, `'`, or newlines (got {s:?})"
+ );
+}
+
+/// Small CLI hint panel: one border and title; each line is hover-highlighted and copies on click.
+pub(super) fn cli_panel<I: AsRef<str>>(cmds: &[I]) -> Markup {
+ if cmds.is_empty() {
+ return html! {};
+ }
+ for cmd in cmds {
+ assert_cli_panel_cmd_js_single_quote_safe(cmd.as_ref());
+ }
html! {
div class="cli-panel" {
span class="cli-panel-label muted" { "cli" }
- code class="cli-panel-cmd" { (cmd) }
- button
- class="cli-panel-copy"
- title="Copy to clipboard"
- onclick=(format!(r#"navigator.clipboard.writeText('{}'); this.textContent='✓'; setTimeout(() => this.textContent='copy', 2000);"#, cmd.replace("'", "\\'")))
- {
- "copy"
+ div class="cli-panel-cmds" {
+ @for cmd in cmds {
+ @let s = cmd.as_ref();
+ button type="button" class="cli-panel-row" title="Copy command" onclick=(format!(
+ r#"navigator.clipboard.writeText('{}');"#,
+ s
+ )) {
+ code class="cli-panel-cmd" { (s) }
+ }
+ }
}
}
}
diff --git a/server/src/html/search.rs b/server/src/html/search.rs
index 28ceaa53c3fcac6777311535e95fb771b19438f5..43e6ebf36cf0fe72c96f0f9d850bea51ac094c43 100644
--- a/server/src/html/search.rs
+++ b/server/src/html/search.rs
@@ -418,7 +418,7 @@ pub async fn search_page(
value=(query) autocomplete="off" autofocus;
}
(render_search_results(&results, &query))
- (cli_panel("npx slugsocial search <query>"))
+ (cli_panel(&["npx slugsocial search <query>"]))
},
None,
theme_from_jar(&jar),
diff --git a/server/static/theme_default.css b/server/static/theme_default.css
index e024a5c8b74139ae8be37b2a1bd17e4c3abe9324..764b66c8208e91e6138b83c534387cf43c85b8b5 100644
--- a/server/static/theme_default.css
+++ b/server/static/theme_default.css
@@ -610,7 +610,7 @@ code {
CLI PANEL — how to view this page from the terminal
---------------------------------------------------------------- */
div.cli-panel {
- align-items: baseline;
+ align-items: flex-start;
background: var(--g1);
border: var(--bv) solid;
border-color: var(--lo) var(--hi) var(--hi) var(--lo); /* inset */
@@ -621,11 +621,34 @@ div.cli-panel {
width: fit-content;
max-width: 100%;
}
+.cli-panel-cmds {
+ display: flex;
+ flex-direction: column;
+ gap: 4px;
+ flex: 1;
+ min-width: 0;
+}
+button.cli-panel-row {
+ background: transparent;
+ border: none;
+ color: inherit;
+ cursor: pointer;
+ display: block;
+ font: inherit;
+ margin: 0;
+ padding: 2px 4px;
+ text-align: left;
+ width: 100%;
+}
+button.cli-panel-row:hover {
+ background: var(--g3);
+}
.cli-panel-label {
font-size: 11px;
letter-spacing: 0.08em;
text-transform: uppercase;
flex-shrink: 0;
+ padding-top: 2px;
}
.cli-panel-cmd {
background: none;
diff --git a/server/static/theme_retro_craft.css b/server/static/theme_retro_craft.css
index 00714575bfa533d1d9c66653b9089642e2b0a6ca..f89ecbc3a18eb9b2b27d1f7764330f6bd6987552 100644
--- a/server/static/theme_retro_craft.css
+++ b/server/static/theme_retro_craft.css
@@ -306,19 +306,41 @@ a.post-nav-btn:hover {
}
div.cli-panel {
- align-items: baseline;
+ align-items: flex-start;
border: 1px dashed var(--line);
display: flex;
- flex-wrap: wrap;
gap: 0.5rem;
margin: 0.65rem 0;
padding: 0.45rem 0.65rem;
}
+.cli-panel-cmds {
+ display: flex;
+ flex-direction: column;
+ gap: 0.25rem;
+ flex: 1;
+ min-width: 0;
+}
+button.cli-panel-row {
+ background: transparent;
+ border: none;
+ color: inherit;
+ cursor: pointer;
+ display: block;
+ font: inherit;
+ margin: 0;
+ padding: 0.1rem 0.2rem;
+ text-align: left;
+ width: 100%;
+}
+button.cli-panel-row:hover {
+ background: color-mix(in srgb, var(--accent) 14%, transparent);
+}
.cli-panel-label {
color: var(--ink-dim);
font-size: 0.72rem;
letter-spacing: 0.12em;
text-transform: uppercase;
+ padding-top: 0.12rem;
}
.cli-panel-cmd {
color: var(--accent);
B — c_f10e7b043e68 (tommy-mor)
message
[7bb7145d] url stuff
diff preview
diff --git a/AGENTS.md b/AGENTS.md
index e60b9ba6012593361ef10e8fdd9439cd9932e09b..babb889d6fbfb1fa7176c9e6b7544ae17b61dd2e 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -58,4 +58,4 @@ Use **tmux** for `cargo run --package sorter2-server` (dev server). Rebuild afte
- First `cargo test` / `cargo build --release` is slow; Clojure smoke test always does a release build.
- `legacy/` and `ideas/` are not part of the workspace build.
-- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/`](server/src/url_rules/) (composable Rust, not a config DSL). After changing canonicalization rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
+- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/graph.rs`](server/src/url_rules/graph.rs): a semantic graph (DFA on host + path, query params in `Context`) with a generic internet fallback for unknown sites. After changing rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
diff --git a/server/src/url_rules/engine.rs b/server/src/url_rules/engine.rs
deleted file mode 100644
index e29b6b48c08deb7bffe031b1e542b1e25a7bef15..0000000000000000000000000000000000000000
--- a/server/src/url_rules/engine.rs
+++ /dev/null
@@ -1,187 +0,0 @@
-//! Composable URL normalization primitives.
-
-use std::collections::HashMap;
-
-use url::Url;
-
-/// Mutable URL view used by rule combinators before serializing to a canonical string.
-#[derive(Debug, Clone)]
-pub struct ParsedUrl {
- pub scheme: String,
- pub host: String,
- pub path_segments: Vec<String>,
- pub query: HashMap<String, String>,
- pub fragment: Option<String>,
-}
-
-impl ParsedUrl {
- pub fn parse(raw: &str) -> Option<Self> {
- let trimmed = raw.trim();
- if trimmed.is_empty() {
- return None;
- }
-
- let with_scheme = if trimmed.contains("://") {
- trimmed.to_string()
- } else if trimmed.starts_with("r/") || trimmed.starts_with("/r/") {
- let rest = trimmed.trim_start_matches('/').trim_start_matches("r/");
- format!("https://reddit.com/r/{rest}")
- } else if trimmed.contains('.') && !trimmed.starts_with('/') {
- format!("https://{trimmed}")
- } else {
- trimmed.to_string()
- };
-
- let url = Url::parse(&with_scheme).ok()?;
- let host = url.host_str()?.to_string();
- let path_segments: Vec<String> = url
- .path_segments()
- .map(|segs| segs.filter(|s| !s.is_empty()).map(str::to_string).collect())
- .unwrap_or_default();
-
- let mut query = HashMap::new();
- for (k, v) in url.query_pairs() {
- query.insert(k.into_owned(), v.into_owned());
- }
-
- Some(Self {
- scheme: url.scheme().to_string(),
- path_segments,
- query,
- fragment: url.fragment().map(str::to_string),
- host,
- })
- }
-
- pub fn with_path_segments(&self, segments: &[String]) -> Self {
- let mut u = self.clone();
- u.path_segments = segments.to_vec();
- u
- }
-
- pub fn to_url(&self) -> Option<Url> {
- let mut url = if self.path_segments.is_empty() {
- Url::parse(&format!("{}://{}", self.scheme, self.host)).ok()?
- } else {
- let path = format!("/{}", self.path_segments.join("/"));
- Url::parse(&format!("{}://{}{}", self.scheme, self.host, path)).ok()?
- };
- if !self.query.is_empty() {
- let mut pairs: Vec<_> = self.query.iter().collect();
- pairs.sort_by(|a, b| a.0.cmp(b.0));
- url.query_pairs_mut().clear();
- for (k, v) in pairs {
- url.query_pairs_mut().append_pair(k, v);
- }
- }
- if let Some(ref frag) = self.fragment {
- url.set_fragment(Some(frag));
- }
- Some(url)
- }
-
- pub fn canonical_string(&self) -> Option<String> {
- let url = self.to_url()?;
- let mut s = url.to_string();
- if self.path_segments.is_empty() {
- s = s.trim_end_matches('/').to_string();
- }
- Some(s)
- }
-}
-
-pub fn force_https(u: &mut ParsedUrl) {
- if u.scheme == "http" {
- u.scheme = "https".to_string();
- }
-}
-
-pub fn drop_fragment(u: &mut ParsedUrl) {
- u.fragment = None;
-}
-
-pub fn strip_www(u: &mut ParsedUrl) {
- if u.host.starts_with("www.") {
- u.host = u.host[4..].to_string();
- }
-}
-
-pub fn lowercase_host(u: &mut ParsedUrl) {
- u.host = u.host.to_ascii_lowercase();
-}
-
-pub fn lowercase_path(u: &mut ParsedUrl) {
- for seg in &mut u.path_segments {
- *seg = seg.to_ascii_lowercase();
- }
-}
-
-pub fn clear_query(u: &mut ParsedUrl) {
- u.query.clear();
-}
-
-pub fn keep_only_query(u: &mut ParsedUrl, keys: &[&str]) {
- u.query
- .retain(|k, _| keys.iter().any(|want| want == &k.as_str()));
-}
-
-pub fn strip_tracking_params(u: &mut ParsedUrl) {
- u.query.retain(|k, _| {
- let lower = k.to_ascii_lowercase();
- !(lower.starts_with("utm_")
- || matches!(
- lower.as_str(),
- "fbclid" | "gclid" | "ref" | "ref_src" | "ref_source" | "mc_cid" | "mc_eid"
- ))
- });
-}
-
-pub fn truncate_after_segment(u: &mut ParsedUrl, name: &str, keep: usize) {
- if let Some(i) = u.path_segments.iter().position(|s| s == name) {
- let end = (i + 1 + keep).min(u.path_segments.len());
- u.path_segments.truncate(end);
- }
-}
-
-pub fn drop_listing_suffix(u: &mut ParsedUrl, suffixes: &[&str]) {
- if u.path_segments.len() >= 3 && u.path_segments.first().map(String::as_str) == Some("r") {
- if let Some(last) = u.path_segments.last() {
- if suffixes.iter().any(|s| *s == last.as_str()) {
- u.path_segments.pop();
- }
- }
- }
-}
-
-pub fn normalize_reddit_host(u: &mut ParsedUrl) {
- if matches!(
- u.host.as_str(),
- "old.reddit.com" | "new.reddit.com" | "www.reddit.com"
- ) {
- u.host = "reddit.com".to_string();
- }
-}
-
-pub fn rewrite_youtu_be(u: &mut ParsedUrl) {
- if u.host == "youtu.be" && u.path_segments.len() == 1 {
- let id = u.path_segments[0].clone();
- u.host = "youtube.com".to_string();
- u.path_segments = vec!["watch".to_string()];
- u.query.insert("v".to_string(), id);
- }
-}
-
-pub fn rewrite_youtube_shorts(u: &mut ParsedUrl) {
- if u.host == "youtube.com" && u.path_segments.first().map(String::as_str) == Some("shorts") {
- if let Some(id) = u.path_segments.get(1).cloned() {
- u.path_segments = vec!["watch".to_string()];
- u.query.insert("v".to_string(), id);
- }
- }
-}
-
-pub fn normalize_youtube_host(u: &mut ParsedUrl) {
- if matches!(u.host.as_str(), "m.youtube.com" | "www.youtube.com") {
- u.host = "youtube.com".to_string();
- }
-}
diff --git a/server/src/url_rules/mod.rs b/server/src/url_rules/mod.rs
index 03d53bd3e82d704a01ba3fd8dd02b7d31422c0de..9e1445346ce77a49dd6a7e7713bf9c57aef353cc 100644
--- a/server/src/url_rules/mod.rs
+++ b/server/src/url_rules/mod.rs
@@ -1,8 +1,12 @@
-//! URL canonicalization and hierarchy rules for [`crate::path_types::ItemId`].
+//! URL canonicalization and hierarchy via a semantic graph (DFA + generic fallback).
-mod engine;
+mod graph;
+mod parse;
mod registry;
+#[cfg(test)]
+mod registry_tests;
+
pub use registry::{
canonicalize_raw, looks_like_url, navigable_breadcrumbs, parent_url, resolve_id, CanonicalResult,
};
diff --git a/server/src/url_rules/registry.rs b/server/src/url_rules/registry.rs
index 14514e9af8385fb2b9b2f35eb9ee14d453d4b97c..8e6c012ea1fc74b864307bdacdf5a0f5db5259fc 100644
--- a/server/src/url_rules/registry.rs
+++ b/server/src/url_rules/registry.rs
@@ -1,12 +1,7 @@
-//! Per-domain canonicalization and hierarchy rules.
+//! Public API: canonical identity and hierarchy via the URL graph.
-use std::collections::HashSet;
-
-use super::engine::{
- clear_query, drop_fragment, drop_listing_suffix, force_https, keep_only_query, lowercase_host,
- lowercase_path, normalize_reddit_host, normalize_youtube_host, rewrite_youtu_be,
- rewrite_youtube_shorts, strip_tracking_params, strip_www, truncate_after_segment, ParsedUrl,
-};
+use super::graph::graph;
+use super::parse::UrlParts;
/// Result of canonicalizing a raw URL string.
#[derive(Debug, Clone, PartialEq, Eq)]
@@ -16,71 +11,16 @@ pub struct CanonicalResult {
pub alias_of: Option<String>,
}
-fn apply_global(u: &mut ParsedUrl) {
- force_https(u);
- drop_fragment(u);
- strip_www(u);
- lowercase_host(u);
- strip_tracking_params(u);
-}
-
-fn normalize_reddit(u: &mut ParsedUrl) {
- normalize_reddit_host(u);
- lowercase_path(u);
- truncate_after_segment(u, "comments", 1);
- drop_listing_suffix(u, &["hot", "top", "new", "rising", "controversial"]);
- clear_query(u);
-}
-
-fn normalize_youtube(u: &mut ParsedUrl) {
- rewrite_youtu_be(u);
- normalize_youtube_host(u);
- rewrite_youtube_shorts(u);
- keep_only_query(u, &["v", "list"]);
-}
-
-fn normalize_default(_u: &mut ParsedUrl) {
- // Global rules only.
-}
-
-fn domain_key(host: &str) -> &'static str {
- if host == "reddit.com" || host.ends_with(".reddit.com") {
- "reddit.com"
- } else if host == "youtube.com" || host == "youtu.be" {
- "youtube.com"
- } else {
- "default"
- }
-}
-
-fn normalize_for_host(u: &mut ParsedUrl) {
- apply_global(u);
- match domain_key(&u.host) {
- "reddit.com" => normalize_reddit(u),
- "youtube.com" => normalize_youtube(u),
- _ => normalize_default(u),
- }
-}
-
-/// Structural path segments that must not become standalone tree nodes when more path follows.
-fn structural_trailing(host: &str) -> &'static [&'static str] {
- match domain_key(host) {
- "reddit.com" => &["comments"],
- _ => &[],
- }
-}
-
/// Canonicalize a raw URL. Returns `None` if the input is not URL-like.
pub fn canonicalize_raw(raw: &str) -> Option<CanonicalResult> {
let trimmed = raw.trim();
if trimmed.is_empty() {
return None;
}
- let mut u = ParsedUrl::parse(trimmed)?;
- let input_snapshot = u.canonical_string()?;
- normalize_for_host(&mut u);
- let canonical = u.canonical_string()?;
- let alias_of = if input_snapshot != canonical {
+ let parts = UrlParts::parse(trimmed)?;
+ let g = graph();
+ let canonical = g.resolve_canonical(&parts)?;
+ let alias_of = if trimmed != canonical {
Some(trimmed.to_string())
} else {
None
@@ -98,35 +38,14 @@ pub fn resolve_id(raw: &str) -> Option<String> {
/// Navigable ancestor URLs from domain root up to and including `canonical` (full URLs).
pub fn navigable_breadcrumbs(canonical: &str) -> Vec<String> {
- let Some(u) = ParsedUrl::parse(canonical) else {
- return vec![canonical.to_string()];
+ let parts = match UrlParts::parse(canonical) {
+ Some(p) => p,
+ None => return vec![canonical.to_string()],
};
- let structural: HashSet<&str> = structural_trailing(&u.host).iter().copied().collect();
- let n = u.path_segments.len();
- let mut out = Vec::new();
-
- // Domain root (no path segments).
- if let Some(base) = u.with_path_segments(&[]).canonical_string() {
- out.push(base);
- }
-
- for i in 0..n {
- let segs: Vec<String> = u.path_segments[..=i].to_vec();
- let is_last = i == n - 1;
- let seg = u.path_segments[i].as_str();
- if structural.contains(seg) && !is_last {
- continue;
- }
- if let Some(url) = u.with_path_segments(&segs).canonical_string() {
- if out.last() != Some(&url) {
-
… preview truncated; 3,144 characters omittedHardlinks — judgments / attempts / prompt
judgments
attempts
Prompt text is loaded only by the download route.