constitution · epochs · watch · epoch 3

comparison

c_ebc883667d61 (tommy-mor) vs c_f10e7b043e68 (tommy-mor)

download prompt · raw event · cmp_428417f541ad46

council reasoning

~anthropic/claude-sonnet-latest · winner B · 6:4 · permalink

A is a purely mechanical cleanup that removes a leftover duplicate forum.rs to fix module-resolution ambiguity — necessary but trivial housekeeping with no design change. B finalizes a real architectural shift in URL canonicalization, replacing the ad-hoc per-domain rule engine (engine.rs with hardcoded reddit/youtube functions) with a more general graph/DFA-based resolver (graph.rs/parse.rs) plus updated docs, which is a more substantive and lasting improvement to core logic even though some test consolidation obscures full visibility of the new implementation.

~x-ai/grok-latest · winner B · 2:5 · permalink

B replaces the old composable ParsedUrl/engine normalizers and domain_key match arms with a graph-based canonicalization API (registry thin wrapper over graph/parse, docs and tests re-homed), which is lasting core ItemId design. A mostly deletes the leftover monolithic forum.rs and redistributes the same forum HTML into forum/* modules plus a small timeago import fix—necessary structure cleanup, but little new behavior or design.

openai/gpt-chat-latest · winner A · 5:2 · permalink

Side A resolves a real module-structure issue by removing the duplicate `server/src/html/forum.rs`, adding `forum/mod.rs`, and making `forum/` the single module root while preserving behavior through extracted modules like `feed.rs`, `views.rs`, and `thread_morph.rs`. Side B replaces the URL canonicalization engine with a new graph-based architecture and updates the registry API, but the visible patch largely swaps implementations and removes in-file tests without showing the new core logic, making its lasting functional impact less directly evident from the diff.

sides

A — c_ebc883667d61 (tommy-mor)

message

[21b57b50] fix(html): drop duplicate forum.rs; use forum/ as sole module root

Removes the leftover monolithic file so mod forum resolves to forum/mod.rs
without ambiguity. Keeps feed, views, and other split modules as the source.

Made-with: Cursor

diff preview

diff --git a/server/src/html/forum.rs b/server/src/html/forum.rs
deleted file mode 100644
index 5ad8dfc84dd735d589432e2c613ff687a75f2e61..0000000000000000000000000000000000000000
--- a/server/src/html/forum.rs
+++ /dev/null
@@ -1,1405 +0,0 @@
-use axum::{
-    extract::{Path, Query, State},
-    http::{HeaderMap, StatusCode, Uri},
-    response::{Html, IntoResponse},
-};
-use axum_extra::extract::cookie::CookieJar;
-use maud::{html, Markup};
-use serde::Deserialize;
-
-use crate::{
-    api::optional_principal,
-    canonical_path::{canonicalize_item, canonicalize_tag},
-    events::ThreadCapability,
-    form_template::template_json_compact,
-    identity::parse_username,
-    reducer::{scope_from_room_wire, ReducerState, ScopeId},
-    state::AppState,
-    timeago,
-};
-use serde_json::json;
-
-use super::js_string_literal;
-use super::ui_action::{HtmlUiAction, UI_RPC_FIELD};
-
-use super::{
-    bc_segment, bc_threads, cli_panel, layout, now_ms, profile_href, recency_class,
-    render_linkified_with_embeds_in_scope, theme_from_jar, theme_next_from_uri, JsBuilder,
-};
-
-#[derive(Clone)]
-struct ThreadRow {
-    tag: String,
-    subtitle: Option<String>,
-    last_ts: i64,
-    ingests: usize,
-}
-
-#[derive(Clone)]
-struct RoomMemberRow {
-    username: String,
-    capabilities: Vec<&'static str>,
-}
-
-/// URL helpers for public `/t/…` and private room threads `/r/{short}/{slug}/t/…`.
-#[derive(Clone)]
-pub struct ThreadNav {
-    pub room_wire: String,
-    scope: ScopeId,
-    room_path: String,
-    thread_path_prefix: String,
-    garden_path_prefix: String,
-}
-
-impl ThreadNav {
-    pub(crate) fn public() -> Self {
-        Self {
-            room_wire: "public".into(),
-            scope: ScopeId::Public,
-            room_path: "/t".into(),
-            thread_path_prefix: "/t".into(),
-            garden_path_prefix: "/~".into(),
-        }
-    }
-
-    /// `room_id` wire form `shortid/slug`.
-    pub(crate) fn from_room_id(room_id: &str) -> Option<Self> {
-        let (short, slug) = room_id.split_once('/')?;
-        if short.is_empty() || slug.is_empty() {
-            return None;
-        }
-        Some(Self {
-            room_wire: room_id.to_string(),
-            scope: ScopeId::Room(room_id.to_string()),
-            room_path: format!("/r/{short}/{slug}"),
-            thread_path_prefix: format!("/r/{short}/{slug}/t"),
-            garden_path_prefix: format!("/r/{short}/{slug}/~"),
-        })
-    }
-
-    pub(crate) fn scope(&self) -> ScopeId {
-        self.scope.clone()
-    }
-
-    pub(crate) fn room_url(&self) -> &str {
-        &self.room_path
-    }
-
-    pub(crate) fn thread_url(&self, tag: &str) -> String {
-        format!("{}/{}", self.thread_path_prefix, tag)
-    }
-
-    pub(crate) fn garden_root_url(&self) -> &str {
-        &self.garden_path_prefix
-    }
-
-    pub(crate) fn garden_item_url(&self, item: &str) -> String {
-        if let Some(tail) = crate::path_types::CanonicalItemUrl::parse(item)
-            .and_then(|c| c.tilde_tail().map(str::to_owned))
-        {
-            format!("{}/{}", self.garden_path_prefix, tail)
-        } else {
-            format!("{}/{}", self.garden_path_prefix, canonicalize_item(item))
-        }
-    }
-
-    fn thread_page_url(&self, tag: &str, offset: usize) -> String {
-        let base = self.thread_url(tag);
-        if offset == 0 {
-            base
-        } else {
-            format!("{base}?offset={offset}")
-        }
-    }
-
-    fn post_url(&self, tag: &str, idx: usize) -> String {
-        format!("{}/{}/{}", self.thread_path_prefix, tag, idx)
-    }
-}
-
-/// `POST /ui` + `__rpc__` from an inline link (`onclick`); same-origin credentials as other morph actions.
-fn thread_ui_fetch_onclick(rpc_compact_json: &str) -> String {
-    format!(
-        "fetch('/ui',{{method:'POST',headers:{{'Content-Type':'application/x-www-form-urlencoded'}},body:new URLSearchParams({{__rpc__:{}}}).toString(),credentials:'same-origin'}}).then(r=>r.text()).then(eval);return false",
-        js_string_literal(rpc_compact_json)
-    )
-}
-
-fn thread_nav_for_ingest(ing: &crate::events::Ingest) -> Option<ThreadNav> {
-    let room = ing.room_id.trim();
-    if room.is_empty() || room == "public" {
-        Some(ThreadNav::public())
-    } else {
-        ThreadNav::from_room_id(room)
-    }
-}
-
-fn thread_post_index_in_scope(reduced: &ReducerState, ing: &crate::events::Ingest) -> Option<usize> {
-    let scope = scope_from_room_wire(&ing.room_id);
-    let tag = canonicalize_tag(&ing.thread_tag);
-    reduced
-        .ingests_by_scope_thread
-        .get(&(scope, tag))
-        .and_then(|q| q.iter().rev().position(|id| id == &ing.id))
-}
-
-fn post_header_meta(
-    nav: &ThreadNav,
-    tag: &str,
-    post_idx: usize,
-    principal: &str,
-    ts: i64,
-    now: i64,
-) -> Markup {
-    let post_href = nav.post_url(tag, post_idx);
-    let profile = profile_href(principal);
-    let hover = timeago::rfc3339_utc(ts);
-    let ago = timeago::timeago(now, ts);
-    html! {
-        div class="ingest-meta muted" title=(hover) {
-            a href=(post_href) class="post-num" { "#" (post_idx) }
-            " "
-            a href=(profile) class="post-author" { "@" (principal) }
-            " · "
-            (ago)
-        }
-    }
-}
-
-fn post_header_row(
-    nav: &ThreadNav,
-    tag: &str,
-    post_idx: usize,
-    ing: &crate::events::Ingest,
-    _viewer: Option<&str>,
-    now: i64,
-    show_delete: bool,
-) -> Markup {
-    let meta = post_header_meta(nav, tag, post_idx, &ing.principal, ing.ts, now);
-    html! {
-        div class="ingest-header-row" {
-            (meta)
-            @if show_delete {
-                form class="post-delete-form" method="POST" action="/ui" {
-                    input type="hidden" name=(UI_RPC_FIELD) value=(template_json_compact(&HtmlUiAction::RedactPost { post_id: ing.id.clone() }).unwrap());
-                    button type="submit" class="post-delete-btn" { "delete" }
-                }
-            }
-        }
-    }
-}
-
-fn redacted_header_row(
-    nav: &ThreadNav,
-    tag: &str,
-    post_idx: usize,
-    ing: &crate::events::Ingest,
-    now: i64,
-    expanded: bool,
-) -> Markup {
-    let meta = post_header_meta(nav, tag, post_idx, &ing.principal, ing.ts, now);
-    let rpc_expand = template_json_compact(&json!({
-        "action": "expand_redacted_post",
-        "room": nav.room_wire,
-        "thread_tag": tag,
-        "post_index": post_idx,
-    }))
-    .unwrap();
-    let rpc_collapse = template_json_compact(&json!({
-        "action": "collapse_redacted_post",
-        "room": nav.room_wire,
-        "thread_tag": tag,
-        "post_index": post_idx,
-    }))
-    .unwrap();
-    let onclick_expand = thread_ui_fetch_onclick(&rpc_expand);
-    let onclick_collapse = thread_ui_fetch_onclick(&rpc_collapse);
-    html! {
-        div class="ingest-header-row ingest-tombstone-row" {
-            (meta)
-            span class="post-tombstone-inline muted" {
-                "deleted · "
-                @if expanded {
-                    a href="#" class="hide-deleted-link"
-                      onclick=(onclick_collapse) {
-                        "[hide deleted content]"
-                    }
-                } @else {
-                    a href="#" class="show-deleted-link"
-                      onclick=(onclick_expand) {
-                        "[show deleted content]"
-                    }
-                }
-            }
-        }
-    }
-}
-
-fn ingest_entry_markup(
-    nav: &ThreadNav,
-    tag: &str,
-    post_idx: usize,
-    ing: &crate::events::Ingest,
-    viewer: Option<&str>,
-    now: i64,
-    reduced: &ReducerState,
-) -> Markup {
-    let redacted = reduced.redacted_posts.contains(&ing.id);
-    let show_delete = viewer == Some(ing.principal.as_str()) && !redacted;
-    if redacted {
-        html! {
-            div class="ingest-entry ingest-redacted" data-ingest-id=(ing.id) {
-                (redacted_header_row(nav, tag, post_idx, ing, now, false))
-            }
-        }
-    } else {
-        let truncated = ing.raw.len() > 2000;
-        let display_body = if truncated { &ing.raw[..2000] } else { &ing.raw[..] };
-        html! {
-            div class="ingest-entry" data-ingest-id=(ing.id) {
-                (post_header_row(nav, tag, post_idx, ing, viewer, now, show_delete))
-                (render_linkified_with_embeds_in_scope(display_body, nav.garden_root_url()))
-                @if truncated {
-                    @let rpc_full = template_json_compact(&json!({
-                        "action": "expand_post_full",
-                        "room": nav.room_wire,
-                        "thread_tag": tag,
-                        "post_index": post_idx,
-                    })).unwrap();
-                    @let onclick_full = thread_ui_fetch_onclick(&rpc_full);
-                    a href="#" class="show-full-link"
-                      onclick=(onclick_full) {
-                        "[show full post]"
-                    }
-                }
-            }
-        }
-    }
-}
-
-fn collect_thread_rows_for_scope(reduced: &ReducerState, scope: &ScopeId, now: i64) -> Vec<ThreadRow> {
-    let _ = now;
-    reduced
-        .forum_threads
-        .iter()
-        .filter(|((s, _), _)| s == scope)
-        .map(|((_, tag), thread)| {
-            let ingests = reduced
-                .ingests_by_scope_thread
-                .get(&(scope.clone(), tag.clone()))
-                .map(|q| q.len())
-                .unwrap_or(0);
-            ThreadRow {
-                tag: tag.clone(),
-                subtitle: None,
-                last_ts: thread.last_activity_ts,
-                ingests,
-            }
-        })
-        .collect()
-}
-
-fn rooms_for_user(reduced: &ReducerState, username: &str) -> Vec<String> {
-    let mut v: Vec<String> = reduced
-        .grants
-        .iter()
-        .filter(|(rid, m)| reduced.rooms.contains(*rid) && m.contains_key(username))
-        .map(|(rid, _)| rid.clone())
-        .collect();
-    v.sort();
-    v
-}
-
-pub(crate) fn user_can_view_room(reduced: &ReducerState, room_id: &str, username: Option<&str>) -> bool {
-    if !reduced.rooms.contains(room_id) {
-        return false;
-    }
-    let Some(u) = username else {
-        return false;
-    };
-    reduced.user_has_cap(room_id, u, ThreadCapability::View)
-}
-
-pub(crate) fn user_can_post_room(reduced: &ReducerState, room_id: &str, username: &str) -> bool {
-    reduced.user_has_cap(room_id, username, ThreadCapability::Post)
-}
-
-fn capability_label(cap: ThreadCapability) -> &'static str {
-    match cap {
-        ThreadCapability::View => "view",
-        ThreadCapability::Post => "post",
-        ThreadCapability::Vote => "vote",
-        ThreadCapability::AddItem => "add_item",
-        ThreadCapability::Manage => "manage",
-    }
-}
-
-fn room_members_for_room(reduced: &ReducerState, room_id: &str) -> Vec<RoomMemberRow> {
-    let mut rows: Vec<RoomMemberRow> = reduced
-        .grants
-        .get(room_id)
-        .into_iter()
-        .flat_map(|members| members.iter())
-        .map(|(username, caps)| {
-            let mut ordered = Vec::new();
-            for cap in [
-                ThreadCapability::View,
-                ThreadCapability::Post,
-                ThreadCapability::Vote,
-                ThreadCapability::AddItem,
-                ThreadCapability::Manage,
-            ] {
-                if caps.contains(&cap) {
-                    ordered.push(capability_label(cap));
-                }
-            }
-            RoomMemberRow {
-                username: username.clone(),
-                capabilities: ordered,
-            }
-        })
-        .collect();
-    rows.sort_by(|a, b| a.username.cmp(&b.username));
-    rows
-}
-
-fn room_members_inner(members: &[RoomMemberRow]) -> Markup {
-    html! {
-        h3 { "members

… preview truncated; 77,260 characters omitted

download full diff A

B — c_f10e7b043e68 (tommy-mor)

message

[7bb7145d] url stuff

diff preview

diff --git a/AGENTS.md b/AGENTS.md
index e60b9ba6012593361ef10e8fdd9439cd9932e09b..babb889d6fbfb1fa7176c9e6b7544ae17b61dd2e 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -58,4 +58,4 @@ Use **tmux** for `cargo run --package sorter2-server` (dev server). Rebuild afte
 
 - First `cargo test` / `cargo build --release` is slow; Clojure smoke test always does a release build.
 - `legacy/` and `ideas/` are not part of the workspace build.
-- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/`](server/src/url_rules/) (composable Rust, not a config DSL). After changing canonicalization rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
+- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/graph.rs`](server/src/url_rules/graph.rs): a semantic graph (DFA on host + path, query params in `Context`) with a generic internet fallback for unknown sites. After changing rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
diff --git a/server/src/url_rules/engine.rs b/server/src/url_rules/engine.rs
deleted file mode 100644
index e29b6b48c08deb7bffe031b1e542b1e25a7bef15..0000000000000000000000000000000000000000
--- a/server/src/url_rules/engine.rs
+++ /dev/null
@@ -1,187 +0,0 @@
-//! Composable URL normalization primitives.
-
-use std::collections::HashMap;
-
-use url::Url;
-
-/// Mutable URL view used by rule combinators before serializing to a canonical string.
-#[derive(Debug, Clone)]
-pub struct ParsedUrl {
-    pub scheme: String,
-    pub host: String,
-    pub path_segments: Vec<String>,
-    pub query: HashMap<String, String>,
-    pub fragment: Option<String>,
-}
-
-impl ParsedUrl {
-    pub fn parse(raw: &str) -> Option<Self> {
-        let trimmed = raw.trim();
-        if trimmed.is_empty() {
-            return None;
-        }
-
-        let with_scheme = if trimmed.contains("://") {
-            trimmed.to_string()
-        } else if trimmed.starts_with("r/") || trimmed.starts_with("/r/") {
-            let rest = trimmed.trim_start_matches('/').trim_start_matches("r/");
-            format!("https://reddit.com/r/{rest}")
-        } else if trimmed.contains('.') && !trimmed.starts_with('/') {
-            format!("https://{trimmed}")
-        } else {
-            trimmed.to_string()
-        };
-
-        let url = Url::parse(&with_scheme).ok()?;
-        let host = url.host_str()?.to_string();
-        let path_segments: Vec<String> = url
-            .path_segments()
-            .map(|segs| segs.filter(|s| !s.is_empty()).map(str::to_string).collect())
-            .unwrap_or_default();
-
-        let mut query = HashMap::new();
-        for (k, v) in url.query_pairs() {
-            query.insert(k.into_owned(), v.into_owned());
-        }
-
-        Some(Self {
-            scheme: url.scheme().to_string(),
-            path_segments,
-            query,
-            fragment: url.fragment().map(str::to_string),
-            host,
-        })
-    }
-
-    pub fn with_path_segments(&self, segments: &[String]) -> Self {
-        let mut u = self.clone();
-        u.path_segments = segments.to_vec();
-        u
-    }
-
-    pub fn to_url(&self) -> Option<Url> {
-        let mut url = if self.path_segments.is_empty() {
-            Url::parse(&format!("{}://{}", self.scheme, self.host)).ok()?
-        } else {
-            let path = format!("/{}", self.path_segments.join("/"));
-            Url::parse(&format!("{}://{}{}", self.scheme, self.host, path)).ok()?
-        };
-        if !self.query.is_empty() {
-            let mut pairs: Vec<_> = self.query.iter().collect();
-            pairs.sort_by(|a, b| a.0.cmp(b.0));
-            url.query_pairs_mut().clear();
-            for (k, v) in pairs {
-                url.query_pairs_mut().append_pair(k, v);
-            }
-        }
-        if let Some(ref frag) = self.fragment {
-            url.set_fragment(Some(frag));
-        }
-        Some(url)
-    }
-
-    pub fn canonical_string(&self) -> Option<String> {
-        let url = self.to_url()?;
-        let mut s = url.to_string();
-        if self.path_segments.is_empty() {
-            s = s.trim_end_matches('/').to_string();
-        }
-        Some(s)
-    }
-}
-
-pub fn force_https(u: &mut ParsedUrl) {
-    if u.scheme == "http" {
-        u.scheme = "https".to_string();
-    }
-}
-
-pub fn drop_fragment(u: &mut ParsedUrl) {
-    u.fragment = None;
-}
-
-pub fn strip_www(u: &mut ParsedUrl) {
-    if u.host.starts_with("www.") {
-        u.host = u.host[4..].to_string();
-    }
-}
-
-pub fn lowercase_host(u: &mut ParsedUrl) {
-    u.host = u.host.to_ascii_lowercase();
-}
-
-pub fn lowercase_path(u: &mut ParsedUrl) {
-    for seg in &mut u.path_segments {
-        *seg = seg.to_ascii_lowercase();
-    }
-}
-
-pub fn clear_query(u: &mut ParsedUrl) {
-    u.query.clear();
-}
-
-pub fn keep_only_query(u: &mut ParsedUrl, keys: &[&str]) {
-    u.query
-        .retain(|k, _| keys.iter().any(|want| want == &k.as_str()));
-}
-
-pub fn strip_tracking_params(u: &mut ParsedUrl) {
-    u.query.retain(|k, _| {
-        let lower = k.to_ascii_lowercase();
-        !(lower.starts_with("utm_")
-            || matches!(
-                lower.as_str(),
-                "fbclid" | "gclid" | "ref" | "ref_src" | "ref_source" | "mc_cid" | "mc_eid"
-            ))
-    });
-}
-
-pub fn truncate_after_segment(u: &mut ParsedUrl, name: &str, keep: usize) {
-    if let Some(i) = u.path_segments.iter().position(|s| s == name) {
-        let end = (i + 1 + keep).min(u.path_segments.len());
-        u.path_segments.truncate(end);
-    }
-}
-
-pub fn drop_listing_suffix(u: &mut ParsedUrl, suffixes: &[&str]) {
-    if u.path_segments.len() >= 3 && u.path_segments.first().map(String::as_str) == Some("r") {
-        if let Some(last) = u.path_segments.last() {
-            if suffixes.iter().any(|s| *s == last.as_str()) {
-                u.path_segments.pop();
-            }
-        }
-    }
-}
-
-pub fn normalize_reddit_host(u: &mut ParsedUrl) {
-    if matches!(
-        u.host.as_str(),
-        "old.reddit.com" | "new.reddit.com" | "www.reddit.com"
-    ) {
-        u.host = "reddit.com".to_string();
-    }
-}
-
-pub fn rewrite_youtu_be(u: &mut ParsedUrl) {
-    if u.host == "youtu.be" && u.path_segments.len() == 1 {
-        let id = u.path_segments[0].clone();
-        u.host = "youtube.com".to_string();
-        u.path_segments = vec!["watch".to_string()];
-        u.query.insert("v".to_string(), id);
-    }
-}
-
-pub fn rewrite_youtube_shorts(u: &mut ParsedUrl) {
-    if u.host == "youtube.com" && u.path_segments.first().map(String::as_str) == Some("shorts") {
-        if let Some(id) = u.path_segments.get(1).cloned() {
-            u.path_segments = vec!["watch".to_string()];
-            u.query.insert("v".to_string(), id);
-        }
-    }
-}
-
-pub fn normalize_youtube_host(u: &mut ParsedUrl) {
-    if matches!(u.host.as_str(), "m.youtube.com" | "www.youtube.com") {
-        u.host = "youtube.com".to_string();
-    }
-}
diff --git a/server/src/url_rules/mod.rs b/server/src/url_rules/mod.rs
index 03d53bd3e82d704a01ba3fd8dd02b7d31422c0de..9e1445346ce77a49dd6a7e7713bf9c57aef353cc 100644
--- a/server/src/url_rules/mod.rs
+++ b/server/src/url_rules/mod.rs
@@ -1,8 +1,12 @@
-//! URL canonicalization and hierarchy rules for [`crate::path_types::ItemId`].
+//! URL canonicalization and hierarchy via a semantic graph (DFA + generic fallback).
 
-mod engine;
+mod graph;
+mod parse;
 mod registry;
 
+#[cfg(test)]
+mod registry_tests;
+
 pub use registry::{
     canonicalize_raw, looks_like_url, navigable_breadcrumbs, parent_url, resolve_id, CanonicalResult,
 };
diff --git a/server/src/url_rules/registry.rs b/server/src/url_rules/registry.rs
index 14514e9af8385fb2b9b2f35eb9ee14d453d4b97c..8e6c012ea1fc74b864307bdacdf5a0f5db5259fc 100644
--- a/server/src/url_rules/registry.rs
+++ b/server/src/url_rules/registry.rs
@@ -1,12 +1,7 @@
-//! Per-domain canonicalization and hierarchy rules.
+//! Public API: canonical identity and hierarchy via the URL graph.
 
-use std::collections::HashSet;
-
-use super::engine::{
-    clear_query, drop_fragment, drop_listing_suffix, force_https, keep_only_query, lowercase_host,
-    lowercase_path, normalize_reddit_host, normalize_youtube_host, rewrite_youtu_be,
-    rewrite_youtube_shorts, strip_tracking_params, strip_www, truncate_after_segment, ParsedUrl,
-};
+use super::graph::graph;
+use super::parse::UrlParts;
 
 /// Result of canonicalizing a raw URL string.
 #[derive(Debug, Clone, PartialEq, Eq)]
@@ -16,71 +11,16 @@ pub struct CanonicalResult {
     pub alias_of: Option<String>,
 }
 
-fn apply_global(u: &mut ParsedUrl) {
-    force_https(u);
-    drop_fragment(u);
-    strip_www(u);
-    lowercase_host(u);
-    strip_tracking_params(u);
-}
-
-fn normalize_reddit(u: &mut ParsedUrl) {
-    normalize_reddit_host(u);
-    lowercase_path(u);
-    truncate_after_segment(u, "comments", 1);
-    drop_listing_suffix(u, &["hot", "top", "new", "rising", "controversial"]);
-    clear_query(u);
-}
-
-fn normalize_youtube(u: &mut ParsedUrl) {
-    rewrite_youtu_be(u);
-    normalize_youtube_host(u);
-    rewrite_youtube_shorts(u);
-    keep_only_query(u, &["v", "list"]);
-}
-
-fn normalize_default(_u: &mut ParsedUrl) {
-    // Global rules only.
-}
-
-fn domain_key(host: &str) -> &'static str {
-    if host == "reddit.com" || host.ends_with(".reddit.com") {
-        "reddit.com"
-    } else if host == "youtube.com" || host == "youtu.be" {
-        "youtube.com"
-    } else {
-        "default"
-    }
-}
-
-fn normalize_for_host(u: &mut ParsedUrl) {
-    apply_global(u);
-    match domain_key(&u.host) {
-        "reddit.com" => normalize_reddit(u),
-        "youtube.com" => normalize_youtube(u),
-        _ => normalize_default(u),
-    }
-}
-
-/// Structural path segments that must not become standalone tree nodes when more path follows.
-fn structural_trailing(host: &str) -> &'static [&'static str] {
-    match domain_key(host) {
-        "reddit.com" => &["comments"],
-        _ => &[],
-    }
-}
-
 /// Canonicalize a raw URL. Returns `None` if the input is not URL-like.
 pub fn canonicalize_raw(raw: &str) -> Option<CanonicalResult> {
     let trimmed = raw.trim();
     if trimmed.is_empty() {
         return None;
     }
-    let mut u = ParsedUrl::parse(trimmed)?;
-    let input_snapshot = u.canonical_string()?;
-    normalize_for_host(&mut u);
-    let canonical = u.canonical_string()?;
-    let alias_of = if input_snapshot != canonical {
+    let parts = UrlParts::parse(trimmed)?;
+    let g = graph();
+    let canonical = g.resolve_canonical(&parts)?;
+    let alias_of = if trimmed != canonical {
         Some(trimmed.to_string())
     } else {
         None
@@ -98,35 +38,14 @@ pub fn resolve_id(raw: &str) -> Option<String> {
 
 /// Navigable ancestor URLs from domain root up to and including `canonical` (full URLs).
 pub fn navigable_breadcrumbs(canonical: &str) -> Vec<String> {
-    let Some(u) = ParsedUrl::parse(canonical) else {
-        return vec![canonical.to_string()];
+    let parts = match UrlParts::parse(canonical) {
+        Some(p) => p,
+        None => return vec![canonical.to_string()],
     };
-    let structural: HashSet<&str> = structural_trailing(&u.host).iter().copied().collect();
-    let n = u.path_segments.len();
-    let mut out = Vec::new();
-
-    // Domain root (no path segments).
-    if let Some(base) = u.with_path_segments(&[]).canonical_string() {
-        out.push(base);
-    }
-
-    for i in 0..n {
-        let segs: Vec<String> = u.path_segments[..=i].to_vec();
-        let is_last = i == n - 1;
-        let seg = u.path_segments[i].as_str();
-        if structural.contains(seg) && !is_last {
-            continue;
-        }
-        if let Some(url) = u.with_path_segments(&segs).canonical_string() {
-            if out.last() != Some(&url) {
- 

… preview truncated; 3,144 characters omitted

download full diff B

Hardlinks — judgments / attempts / prompt

prompt download

judgments

attempts

Prompt text is loaded only by the download route.