constitution · epochs · watch · epoch 3

comparison

c_597d3f736194 (tommy-mor) vs c_f10e7b043e68 (tommy-mor)

download prompt · raw event · cmp_28f36cdea44ad8

council reasoning

~anthropic/claude-sonnet-latest · winner A · 3:1 · permalink

Side A fixes concrete, verifiable bugs (regex split arg, wrong stream for reading POST body, wrong sendResponseHeaders length, missing null-check on token/state, added error handling) that directly restore broken E2E auth tests, with a clear commit message explaining the fix. Side B is a vague 'url stuff' commit that deletes a working engine.rs and its test suite while referencing new graph.rs/parse.rs modules that aren't shown in the diff, leaving the actual replacement logic and whether functionality/tests were preserved unverifiable from the patch itself.

~x-ai/grok-latest · winner B · 2:5 · permalink

B replaces the ad-hoc ParsedUrl combinator stack (engine.rs + per-host normalize_* in registry) with a lasting semantic-graph design for ItemId canonicalization and breadcrumbs—core production identity logic—while A only repairs test doubles (regex split on '=', getRequestBody, 302 length, nil token/state) and Playwright helpers so E2E auth runs again. Test mock fixes are real and necessary, but they do not outlast B’s redesign of the URL rules subsystem.

openai/gpt-chat-latest · winner A · 5:1 · permalink

Side A fixes concrete regressions in the OAuth test infrastructure by correcting query parsing (`str/split` regex), reading POST bodies from `getRequestBody`, handling null tokens/state safely, fixing redirect responses, wrapping handlers to avoid crashes, and updating Playwright test interactions to use real selectors. Side B mostly restructures the URL canonicalization module by replacing the old engine with new graph/parse modules and updating documentation, but the patch shown primarily removes code and redirects APIs without demonstrating the substantive replacement implementation.

sides

A — c_597d3f736194 (tommy-mor)

message

[075d4d37] Fix OAuth test mocks so Clojure E2E auth flows work again.

HttpServer handlers were crashing on query parsing and token POSTs, which broke Playwright login; also read alias/history via real CSS selectors.

Co-authored-by: Cursor <cursoragent@cursor.com>

diff preview

diff --git a/test/auth_login.clj b/test/auth_login.clj
index 6f2f00d7aa7340e63d1ac465b0a2234cec7a983f..aa63b66ccb2471b710ba3d168d0461252b39fc10 100644
--- a/test/auth_login.clj
+++ b/test/auth_login.clj
@@ -14,27 +14,27 @@
 (defn- type-alias! [pg text]
   (page/evaluate pg
                  (.replace
-                  "(() => { const i = document.getElementById('alias-input'); const f = document.getElementById('alias-check-form'); if (!i || !f) return;
+                  "(() => { const i = document.getElementById('alias-input'); const f = document.getElementById('alias-check-form'); if (!i || !f) return Promise.resolve('missing-form');
                     i.value = __TEXT__;
                     const cf = document.getElementById('alias-claim-field'); if (cf) cf.value = i.value;
                     return fetch(f.action, { method: 'POST', credentials: 'same-origin',
                       headers: { 'Content-Type': 'application/x-www-form-urlencoded' },
                       body: new URLSearchParams(new FormData(f)).toString() })
                       .then(function (r) { return r.text(); })
-                      .then(function (t) { eval(t); }); })()"
+                      .then(function (t) { eval(t); return document.getElementById('alias-status')?.textContent || ''; }); })()"
                   "__TEXT__"
-                  (pr-str text)))
-  (Thread/sleep 400))
+                  (pr-str text))))
 
-(defn- element-text [pg test-id]
+(defn- element-text [pg selector]
   (let [raw (page/evaluate pg
-                           (str "document.querySelector('[data-testid=\"" test-id "\"]')?.textContent || ''"))]
+                           (str "document.querySelector(" (pr-str selector) ")?.textContent || ''"))]
     (when (string? raw) (str/trim raw))))
 
 (defn- wait-for-text [pg test-id text timeout-ms]
-  (let [deadline (+ (System/currentTimeMillis) timeout-ms)]
+  (let [deadline (+ (System/currentTimeMillis) timeout-ms)
+        selector (str "[data-testid=\"" test-id "\"]")]
     (loop []
-      (let [got (or (element-text pg test-id) "")]
+      (let [got (or (element-text pg selector) "")]
         (cond
           (= got text) got
           (< (System/currentTimeMillis) deadline) (do (Thread/sleep 200) (recur))
diff --git a/test/support/mock_oauth.clj b/test/support/mock_oauth.clj
index 5ba7e9be3cf6d2226648e9609a09ed45f306930f..333fe08d56cb06bac2a338cad1c73894d54c4495 100644
--- a/test/support/mock_oauth.clj
+++ b/test/support/mock_oauth.clj
@@ -7,7 +7,7 @@
 (defn- query-param [query key]
   (when query
     (some (fn [pair]
-            (let [[k v] (str/split pair "=" 2)]
+            (let [[k v] (str/split pair #"=" 2)]
               (when (= k key)
                 (URLDecoder/decode (or v "") "UTF-8"))))
           (str/split query #"&"))))
@@ -31,11 +31,11 @@
 
 (defn- send-redirect [^HttpExchange ex location]
   (.set (.getResponseHeaders ex) "Location" location)
-  (.sendResponseHeaders ex 302 -1)
+  (.sendResponseHeaders ex 302 0)
   (.close (.getResponseBody ex)))
 
 (defn- read-form [^HttpExchange ex]
-  (let [body (slurp (.getInputStream ex))]
+  (let [body (slurp (.getRequestBody ex))]
     {:code (query-param body "code")
      :grant (query-param body "grant_type")}))
 
@@ -45,7 +45,7 @@
           (str/replace #"^[Bb]earer " "")))
 
 (defn- parse-token-user [token]
-  (when (str/starts-with? token "mock:")
+  (when (and token (str/starts-with? token "mock:"))
     (parse-mock-user (subs token 5))))
 
 (defn- authorize-redirect [exchange query]
@@ -55,7 +55,7 @@
         user (parse-mock-user mock-user)
         code (str "mock:" (:id user) ":" (:login user))
         loc (str redirect-uri "?code=" (java.net.URLEncoder/encode code "UTF-8")
-                 "&state=" (java.net.URLEncoder/encode state "UTF-8"))]
+                 "&state=" (java.net.URLEncoder/encode (or state "") "UTF-8"))]
     (send-redirect exchange loc)))
 
 (defn start-mock-oauth
@@ -65,49 +65,56 @@
         handler
         (proxy [HttpHandler] []
           (handle [^HttpExchange exchange]
-            (let [uri (.getRequestURI exchange)
-                  path (.getPath uri)
-                  query (.getQuery uri)
-                  method (.getRequestMethod exchange)]
-              (cond
-                ;; GitHub authorize
-                (str/ends-with? path "/login/oauth/authorize")
-                (authorize-redirect exchange query)
+            (try
+              (let [uri (.getRequestURI exchange)
+                    path (.getPath uri)
+                    query (.getQuery uri)
+                    method (.getRequestMethod exchange)]
+                (cond
+                  ;; GitHub authorize
+                  (str/ends-with? path "/login/oauth/authorize")
+                  (authorize-redirect exchange query)
 
-                ;; Reddit authorize
-                (str/ends-with? path "/api/v1/authorize")
-                (authorize-redirect exchange query)
+                  ;; Reddit authorize
+                  (str/ends-with? path "/api/v1/authorize")
+                  (authorize-redirect exchange query)
 
-                ;; GitHub token
-                (and (= method "POST") (str/ends-with? path "/login/oauth/access_token"))
-                (let [code (or (:code (read-form exchange)) "mock:1002:newbie")]
-                  (send-json exchange 200 (str "{\"access_token\":\"" code "\",\"token_type\":\"bearer\"}")))
+                  ;; GitHub token
+                  (and (= method "POST") (str/ends-with? path "/login/oauth/access_token"))
+                  (let [code (or (:code (read-form exchange)) "mock:1002:newbie")]
+                    (send-json exchange 200 (str "{\"access_token\":\"" code "\",\"token_type\":\"bearer\"}")))
 
-                ;; Reddit token (client_credentials for import + authorization_code for login)
-                (and (= method "POST") (str/ends-with? path "/api/v1/access_token"))
-                (let [form (read-form exchange)
-                      grant (or (:grant form) "")
-                      code (or (:code form) "mock:t2_test:redditor")]
-                  (if (= grant "client_credentials")
-                    (send-json exchange 200 "{\"access_token\":\"app-token\",\"token_type\":\"bearer\",\"expires_in\":3600}")
-                    (send-json exchange 200 (str "{\"access_token\":\"" code "\",\"token_type\":\"bearer\",\"expires_in\":3600}"))))
+                  ;; Reddit token (client_credentials for import + authorization_code for login)
+                  (and (= method "POST") (str/ends-with? path "/api/v1/access_token"))
+                  (let [form (read-form exchange)
+                        grant (or (:grant form) "")
+                        code (or (:code form) "mock:t2_test:redditor")]
+                    (if (= grant "client_credentials")
+                      (send-json exchange 200 "{\"access_token\":\"app-token\",\"token_type\":\"bearer\",\"expires_in\":3600}")
+                      (send-json exchange 200 (str "{\"access_token\":\"" code "\",\"token_type\":\"bearer\",\"expires_in\":3600}"))))
 
-                ;; GitHub user
-                (= path "/user")
-                (let [token (bearer-token exchange)
-                      user (or (parse-token-user token) {:id "1002" :login "newbie" :numeric? true})]
-                  (send-json exchange 200
-                             (str "{\"id\":" (:id user) ",\"login\":\"" (:login user) "\"}")))
+                  ;; GitHub user
+                  (= path "/user")
+                  (let [token (bearer-token exchange)
+                        user (or (parse-token-user token) {:id "1002" :login "newbie" :numeric? true})]
+                    (send-json exchange 200
+                               (str "{\"id\":" (:id user) ",\"login\":\"" (:login user) "\"}")))
 
-                ;; Reddit /api/v1/me
-                (str/ends-with? path "/api/v1/me")
-                (let [token (bearer-token exchange)
-                      user (or (parse-token-user token) {:id "t2_test" :login "redditor"})]
-                  (send-json exchange 200
-                             (str "{\"id\":\"" (:id user) "\",\"name\":\"" (:login user) "\"}")))
+                  ;; Reddit /api/v1/me
+                  (str/ends-with? path "/api/v1/me")
+                  (let [token (bearer-token exchange)
+                        user (or (parse-token-user token) {:id "t2_test" :login "redditor"})]
+                    (send-json exchange 200
+                               (str "{\"id\":\"" (:id user) "\",\"name\":\"" (:login user) "\"}")))
 
-                :else
-                (send-json exchange 404 "{\"error\":\"not found\"}")))))]
+                  :else
+                  (send-json exchange 404 "{\"error\":\"not found\"}")))
+              (catch Throwable t
+                (binding [*out* *err*]
+                  (println "mock-oauth handler error:" t))
+                (try
+                  (send-json exchange 500 "{\"error\":\"mock-oauth internal\"}")
+                  (catch Throwable _))))))]
     (.createContext server "/" handler)
     (.setExecutor server nil)
     (.start server)
diff --git a/test/support/mock_reddit.clj b/test/support/mock_reddit.clj
index a630cf0938722193e9af88382d60e777ff371be4..faa27945b6394363914c853627a0bad1e819a5f9 100644
--- a/test/support/mock_reddit.clj
+++ b/test/support/mock_reddit.clj
@@ -12,7 +12,7 @@
 (defn- query-param [query key]
   (when query
     (some (fn [pair]
-            (let [[k v] (str/split pair "=" 2)]
+            (let [[k v] (str/split pair #"=" 2)]
               (when (= k key)
                 (URLDecoder/decode (or v "") "UTF-8"))))
           (str/split query #"&"))))
@@ -34,11 +34,11 @@
 
 (defn- send-redirect [^HttpExchange ex location]
   (.set (.getResponseHeaders ex) "Location" location)
-  (.sendResponseHeaders ex 302 -1)
+  (.sendResponseHeaders ex 302 0)
   (.close (.getResponseBody ex)))
 
 (defn- read-form [^HttpExchange ex]
-  (let [body (slurp (.getInputStream ex))]
+  (let [body (slurp (.getRequestBody ex))]
     {:code (query-param body "code")
      :grant (query-param body "grant_type")}))
 
@@ -48,7 +48,7 @@
           (str/replace #"^[Bb]earer " "")))
 
 (defn- parse-token-user [token]
-  (when (str/starts-with? token "mock:")
+  (when (and token (str/starts-with? token "mock:"))
     (parse-mock-user (subs token 5))))
 
 (defn start-mock-reddit
@@ -72,7 +72,7 @@
                        user (parse-mock-user (query-param query "mock_user"))
                        code (str "mock:" (:id user) ":" (:login user))
                        loc (str redirect-uri "?code=" (java.net.URLEncoder/encode code "UTF-8")
-                                "&state=" (java.net.URLEncoder/encode state "UTF-8"))]
+                                "&state=" (java.net.URLEncoder/encode (or state "") "UTF-8"))]
                    (send-redirect exchange loc))
 
                  (and (= method "POST") (str/ends-with? path "/api/v1/access_token"))

download full diff A

B — c_f10e7b043e68 (tommy-mor)

message

[7bb7145d] url stuff

diff preview

diff --git a/AGENTS.md b/AGENTS.md
index e60b9ba6012593361ef10e8fdd9439cd9932e09b..babb889d6fbfb1fa7176c9e6b7544ae17b61dd2e 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -58,4 +58,4 @@ Use **tmux** for `cargo run --package sorter2-server` (dev server). Rebuild afte
 
 - First `cargo test` / `cargo build --release` is slow; Clojure smoke test always does a release build.
 - `legacy/` and `ideas/` are not part of the workspace build.
-- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/`](server/src/url_rules/) (composable Rust, not a config DSL). After changing canonicalization rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
+- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/graph.rs`](server/src/url_rules/graph.rs): a semantic graph (DFA on host + path, query params in `Context`) with a generic internet fallback for unknown sites. After changing rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`.
diff --git a/server/src/url_rules/engine.rs b/server/src/url_rules/engine.rs
deleted file mode 100644
index e29b6b48c08deb7bffe031b1e542b1e25a7bef15..0000000000000000000000000000000000000000
--- a/server/src/url_rules/engine.rs
+++ /dev/null
@@ -1,187 +0,0 @@
-//! Composable URL normalization primitives.
-
-use std::collections::HashMap;
-
-use url::Url;
-
-/// Mutable URL view used by rule combinators before serializing to a canonical string.
-#[derive(Debug, Clone)]
-pub struct ParsedUrl {
-    pub scheme: String,
-    pub host: String,
-    pub path_segments: Vec<String>,
-    pub query: HashMap<String, String>,
-    pub fragment: Option<String>,
-}
-
-impl ParsedUrl {
-    pub fn parse(raw: &str) -> Option<Self> {
-        let trimmed = raw.trim();
-        if trimmed.is_empty() {
-            return None;
-        }
-
-        let with_scheme = if trimmed.contains("://") {
-            trimmed.to_string()
-        } else if trimmed.starts_with("r/") || trimmed.starts_with("/r/") {
-            let rest = trimmed.trim_start_matches('/').trim_start_matches("r/");
-            format!("https://reddit.com/r/{rest}")
-        } else if trimmed.contains('.') && !trimmed.starts_with('/') {
-            format!("https://{trimmed}")
-        } else {
-            trimmed.to_string()
-        };
-
-        let url = Url::parse(&with_scheme).ok()?;
-        let host = url.host_str()?.to_string();
-        let path_segments: Vec<String> = url
-            .path_segments()
-            .map(|segs| segs.filter(|s| !s.is_empty()).map(str::to_string).collect())
-            .unwrap_or_default();
-
-        let mut query = HashMap::new();
-        for (k, v) in url.query_pairs() {
-            query.insert(k.into_owned(), v.into_owned());
-        }
-
-        Some(Self {
-            scheme: url.scheme().to_string(),
-            path_segments,
-            query,
-            fragment: url.fragment().map(str::to_string),
-            host,
-        })
-    }
-
-    pub fn with_path_segments(&self, segments: &[String]) -> Self {
-        let mut u = self.clone();
-        u.path_segments = segments.to_vec();
-        u
-    }
-
-    pub fn to_url(&self) -> Option<Url> {
-        let mut url = if self.path_segments.is_empty() {
-            Url::parse(&format!("{}://{}", self.scheme, self.host)).ok()?
-        } else {
-            let path = format!("/{}", self.path_segments.join("/"));
-            Url::parse(&format!("{}://{}{}", self.scheme, self.host, path)).ok()?
-        };
-        if !self.query.is_empty() {
-            let mut pairs: Vec<_> = self.query.iter().collect();
-            pairs.sort_by(|a, b| a.0.cmp(b.0));
-            url.query_pairs_mut().clear();
-            for (k, v) in pairs {
-                url.query_pairs_mut().append_pair(k, v);
-            }
-        }
-        if let Some(ref frag) = self.fragment {
-            url.set_fragment(Some(frag));
-        }
-        Some(url)
-    }
-
-    pub fn canonical_string(&self) -> Option<String> {
-        let url = self.to_url()?;
-        let mut s = url.to_string();
-        if self.path_segments.is_empty() {
-            s = s.trim_end_matches('/').to_string();
-        }
-        Some(s)
-    }
-}
-
-pub fn force_https(u: &mut ParsedUrl) {
-    if u.scheme == "http" {
-        u.scheme = "https".to_string();
-    }
-}
-
-pub fn drop_fragment(u: &mut ParsedUrl) {
-    u.fragment = None;
-}
-
-pub fn strip_www(u: &mut ParsedUrl) {
-    if u.host.starts_with("www.") {
-        u.host = u.host[4..].to_string();
-    }
-}
-
-pub fn lowercase_host(u: &mut ParsedUrl) {
-    u.host = u.host.to_ascii_lowercase();
-}
-
-pub fn lowercase_path(u: &mut ParsedUrl) {
-    for seg in &mut u.path_segments {
-        *seg = seg.to_ascii_lowercase();
-    }
-}
-
-pub fn clear_query(u: &mut ParsedUrl) {
-    u.query.clear();
-}
-
-pub fn keep_only_query(u: &mut ParsedUrl, keys: &[&str]) {
-    u.query
-        .retain(|k, _| keys.iter().any(|want| want == &k.as_str()));
-}
-
-pub fn strip_tracking_params(u: &mut ParsedUrl) {
-    u.query.retain(|k, _| {
-        let lower = k.to_ascii_lowercase();
-        !(lower.starts_with("utm_")
-            || matches!(
-                lower.as_str(),
-                "fbclid" | "gclid" | "ref" | "ref_src" | "ref_source" | "mc_cid" | "mc_eid"
-            ))
-    });
-}
-
-pub fn truncate_after_segment(u: &mut ParsedUrl, name: &str, keep: usize) {
-    if let Some(i) = u.path_segments.iter().position(|s| s == name) {
-        let end = (i + 1 + keep).min(u.path_segments.len());
-        u.path_segments.truncate(end);
-    }
-}
-
-pub fn drop_listing_suffix(u: &mut ParsedUrl, suffixes: &[&str]) {
-    if u.path_segments.len() >= 3 && u.path_segments.first().map(String::as_str) == Some("r") {
-        if let Some(last) = u.path_segments.last() {
-            if suffixes.iter().any(|s| *s == last.as_str()) {
-                u.path_segments.pop();
-            }
-        }
-    }
-}
-
-pub fn normalize_reddit_host(u: &mut ParsedUrl) {
-    if matches!(
-        u.host.as_str(),
-        "old.reddit.com" | "new.reddit.com" | "www.reddit.com"
-    ) {
-        u.host = "reddit.com".to_string();
-    }
-}
-
-pub fn rewrite_youtu_be(u: &mut ParsedUrl) {
-    if u.host == "youtu.be" && u.path_segments.len() == 1 {
-        let id = u.path_segments[0].clone();
-        u.host = "youtube.com".to_string();
-        u.path_segments = vec!["watch".to_string()];
-        u.query.insert("v".to_string(), id);
-    }
-}
-
-pub fn rewrite_youtube_shorts(u: &mut ParsedUrl) {
-    if u.host == "youtube.com" && u.path_segments.first().map(String::as_str) == Some("shorts") {
-        if let Some(id) = u.path_segments.get(1).cloned() {
-            u.path_segments = vec!["watch".to_string()];
-            u.query.insert("v".to_string(), id);
-        }
-    }
-}
-
-pub fn normalize_youtube_host(u: &mut ParsedUrl) {
-    if matches!(u.host.as_str(), "m.youtube.com" | "www.youtube.com") {
-        u.host = "youtube.com".to_string();
-    }
-}
diff --git a/server/src/url_rules/mod.rs b/server/src/url_rules/mod.rs
index 03d53bd3e82d704a01ba3fd8dd02b7d31422c0de..9e1445346ce77a49dd6a7e7713bf9c57aef353cc 100644
--- a/server/src/url_rules/mod.rs
+++ b/server/src/url_rules/mod.rs
@@ -1,8 +1,12 @@
-//! URL canonicalization and hierarchy rules for [`crate::path_types::ItemId`].
+//! URL canonicalization and hierarchy via a semantic graph (DFA + generic fallback).
 
-mod engine;
+mod graph;
+mod parse;
 mod registry;
 
+#[cfg(test)]
+mod registry_tests;
+
 pub use registry::{
     canonicalize_raw, looks_like_url, navigable_breadcrumbs, parent_url, resolve_id, CanonicalResult,
 };
diff --git a/server/src/url_rules/registry.rs b/server/src/url_rules/registry.rs
index 14514e9af8385fb2b9b2f35eb9ee14d453d4b97c..8e6c012ea1fc74b864307bdacdf5a0f5db5259fc 100644
--- a/server/src/url_rules/registry.rs
+++ b/server/src/url_rules/registry.rs
@@ -1,12 +1,7 @@
-//! Per-domain canonicalization and hierarchy rules.
+//! Public API: canonical identity and hierarchy via the URL graph.
 
-use std::collections::HashSet;
-
-use super::engine::{
-    clear_query, drop_fragment, drop_listing_suffix, force_https, keep_only_query, lowercase_host,
-    lowercase_path, normalize_reddit_host, normalize_youtube_host, rewrite_youtu_be,
-    rewrite_youtube_shorts, strip_tracking_params, strip_www, truncate_after_segment, ParsedUrl,
-};
+use super::graph::graph;
+use super::parse::UrlParts;
 
 /// Result of canonicalizing a raw URL string.
 #[derive(Debug, Clone, PartialEq, Eq)]
@@ -16,71 +11,16 @@ pub struct CanonicalResult {
     pub alias_of: Option<String>,
 }
 
-fn apply_global(u: &mut ParsedUrl) {
-    force_https(u);
-    drop_fragment(u);
-    strip_www(u);
-    lowercase_host(u);
-    strip_tracking_params(u);
-}
-
-fn normalize_reddit(u: &mut ParsedUrl) {
-    normalize_reddit_host(u);
-    lowercase_path(u);
-    truncate_after_segment(u, "comments", 1);
-    drop_listing_suffix(u, &["hot", "top", "new", "rising", "controversial"]);
-    clear_query(u);
-}
-
-fn normalize_youtube(u: &mut ParsedUrl) {
-    rewrite_youtu_be(u);
-    normalize_youtube_host(u);
-    rewrite_youtube_shorts(u);
-    keep_only_query(u, &["v", "list"]);
-}
-
-fn normalize_default(_u: &mut ParsedUrl) {
-    // Global rules only.
-}
-
-fn domain_key(host: &str) -> &'static str {
-    if host == "reddit.com" || host.ends_with(".reddit.com") {
-        "reddit.com"
-    } else if host == "youtube.com" || host == "youtu.be" {
-        "youtube.com"
-    } else {
-        "default"
-    }
-}
-
-fn normalize_for_host(u: &mut ParsedUrl) {
-    apply_global(u);
-    match domain_key(&u.host) {
-        "reddit.com" => normalize_reddit(u),
-        "youtube.com" => normalize_youtube(u),
-        _ => normalize_default(u),
-    }
-}
-
-/// Structural path segments that must not become standalone tree nodes when more path follows.
-fn structural_trailing(host: &str) -> &'static [&'static str] {
-    match domain_key(host) {
-        "reddit.com" => &["comments"],
-        _ => &[],
-    }
-}
-
 /// Canonicalize a raw URL. Returns `None` if the input is not URL-like.
 pub fn canonicalize_raw(raw: &str) -> Option<CanonicalResult> {
     let trimmed = raw.trim();
     if trimmed.is_empty() {
         return None;
     }
-    let mut u = ParsedUrl::parse(trimmed)?;
-    let input_snapshot = u.canonical_string()?;
-    normalize_for_host(&mut u);
-    let canonical = u.canonical_string()?;
-    let alias_of = if input_snapshot != canonical {
+    let parts = UrlParts::parse(trimmed)?;
+    let g = graph();
+    let canonical = g.resolve_canonical(&parts)?;
+    let alias_of = if trimmed != canonical {
         Some(trimmed.to_string())
     } else {
         None
@@ -98,35 +38,14 @@ pub fn resolve_id(raw: &str) -> Option<String> {
 
 /// Navigable ancestor URLs from domain root up to and including `canonical` (full URLs).
 pub fn navigable_breadcrumbs(canonical: &str) -> Vec<String> {
-    let Some(u) = ParsedUrl::parse(canonical) else {
-        return vec![canonical.to_string()];
+    let parts = match UrlParts::parse(canonical) {
+        Some(p) => p,
+        None => return vec![canonical.to_string()],
     };
-    let structural: HashSet<&str> = structural_trailing(&u.host).iter().copied().collect();
-    let n = u.path_segments.len();
-    let mut out = Vec::new();
-
-    // Domain root (no path segments).
-    if let Some(base) = u.with_path_segments(&[]).canonical_string() {
-        out.push(base);
-    }
-
-    for i in 0..n {
-        let segs: Vec<String> = u.path_segments[..=i].to_vec();
-        let is_last = i == n - 1;
-        let seg = u.path_segments[i].as_str();
-        if structural.contains(seg) && !is_last {
-            continue;
-        }
-        if let Some(url) = u.with_path_segments(&segs).canonical_string() {
-            if out.last() != Some(&url) {
- 

… preview truncated; 3,144 characters omitted

download full diff B

Hardlinks — judgments / attempts / prompt

prompt download

judgments

attempts

Prompt text is loaded only by the download route.