You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [075d4d37] Fix OAuth test mocks so Clojure E2E auth flows work again. HttpServer handlers were crashing on query parsing and token POSTs, which broke Playwright login; also read alias/history via real CSS selectors. Co-authored-by: Cursor Side A — unified diff (full patch): diff --git a/test/auth_login.clj b/test/auth_login.clj index 6f2f00d7aa7340e63d1ac465b0a2234cec7a983f..aa63b66ccb2471b710ba3d168d0461252b39fc10 100644 --- a/test/auth_login.clj +++ b/test/auth_login.clj @@ -14,27 +14,27 @@ (defn- type-alias! [pg text] (page/evaluate pg (.replace - "(() => { const i = document.getElementById('alias-input'); const f = document.getElementById('alias-check-form'); if (!i || !f) return; + "(() => { const i = document.getElementById('alias-input'); const f = document.getElementById('alias-check-form'); if (!i || !f) return Promise.resolve('missing-form'); i.value = __TEXT__; const cf = document.getElementById('alias-claim-field'); if (cf) cf.value = i.value; return fetch(f.action, { method: 'POST', credentials: 'same-origin', headers: { 'Content-Type': 'application/x-www-form-urlencoded' }, body: new URLSearchParams(new FormData(f)).toString() }) .then(function (r) { return r.text(); }) - .then(function (t) { eval(t); }); })()" + .then(function (t) { eval(t); return document.getElementById('alias-status')?.textContent || ''; }); })()" "__TEXT__" - (pr-str text))) - (Thread/sleep 400)) + (pr-str text)))) -(defn- element-text [pg test-id] +(defn- element-text [pg selector] (let [raw (page/evaluate pg - (str "document.querySelector('[data-testid=\"" test-id "\"]')?.textContent || ''"))] + (str "document.querySelector(" (pr-str selector) ")?.textContent || ''"))] (when (string? raw) (str/trim raw)))) (defn- wait-for-text [pg test-id text timeout-ms] - (let [deadline (+ (System/currentTimeMillis) timeout-ms)] + (let [deadline (+ (System/currentTimeMillis) timeout-ms) + selector (str "[data-testid=\"" test-id "\"]")] (loop [] - (let [got (or (element-text pg test-id) "")] + (let [got (or (element-text pg selector) "")] (cond (= got text) got (< (System/currentTimeMillis) deadline) (do (Thread/sleep 200) (recur)) diff --git a/test/support/mock_oauth.clj b/test/support/mock_oauth.clj index 5ba7e9be3cf6d2226648e9609a09ed45f306930f..333fe08d56cb06bac2a338cad1c73894d54c4495 100644 --- a/test/support/mock_oauth.clj +++ b/test/support/mock_oauth.clj @@ -7,7 +7,7 @@ (defn- query-param [query key] (when query (some (fn [pair] - (let [[k v] (str/split pair "=" 2)] + (let [[k v] (str/split pair #"=" 2)] (when (= k key) (URLDecoder/decode (or v "") "UTF-8")))) (str/split query #"&")))) @@ -31,11 +31,11 @@ (defn- send-redirect [^HttpExchange ex location] (.set (.getResponseHeaders ex) "Location" location) - (.sendResponseHeaders ex 302 -1) + (.sendResponseHeaders ex 302 0) (.close (.getResponseBody ex))) (defn- read-form [^HttpExchange ex] - (let [body (slurp (.getInputStream ex))] + (let [body (slurp (.getRequestBody ex))] {:code (query-param body "code") :grant (query-param body "grant_type")})) @@ -45,7 +45,7 @@ (str/replace #"^[Bb]earer " ""))) (defn- parse-token-user [token] - (when (str/starts-with? token "mock:") + (when (and token (str/starts-with? token "mock:")) (parse-mock-user (subs token 5)))) (defn- authorize-redirect [exchange query] @@ -55,7 +55,7 @@ user (parse-mock-user mock-user) code (str "mock:" (:id user) ":" (:login user)) loc (str redirect-uri "?code=" (java.net.URLEncoder/encode code "UTF-8") - "&state=" (java.net.URLEncoder/encode state "UTF-8"))] + "&state=" (java.net.URLEncoder/encode (or state "") "UTF-8"))] (send-redirect exchange loc))) (defn start-mock-oauth @@ -65,49 +65,56 @@ handler (proxy [HttpHandler] [] (handle [^HttpExchange exchange] - (let [uri (.getRequestURI exchange) - path (.getPath uri) - query (.getQuery uri) - method (.getRequestMethod exchange)] - (cond - ;; GitHub authorize - (str/ends-with? path "/login/oauth/authorize") - (authorize-redirect exchange query) + (try + (let [uri (.getRequestURI exchange) + path (.getPath uri) + query (.getQuery uri) + method (.getRequestMethod exchange)] + (cond + ;; GitHub authorize + (str/ends-with? path "/login/oauth/authorize") + (authorize-redirect exchange query) - ;; Reddit authorize - (str/ends-with? path "/api/v1/authorize") - (authorize-redirect exchange query) + ;; Reddit authorize + (str/ends-with? path "/api/v1/authorize") + (authorize-redirect exchange query) - ;; GitHub token - (and (= method "POST") (str/ends-with? path "/login/oauth/access_token")) - (let [code (or (:code (read-form exchange)) "mock:1002:newbie")] - (send-json exchange 200 (str "{\"access_token\":\"" code "\",\"token_type\":\"bearer\"}"))) + ;; GitHub token + (and (= method "POST") (str/ends-with? path "/login/oauth/access_token")) + (let [code (or (:code (read-form exchange)) "mock:1002:newbie")] + (send-json exchange 200 (str "{\"access_token\":\"" code "\",\"token_type\":\"bearer\"}"))) - ;; Reddit token (client_credentials for import + authorization_code for login) - (and (= method "POST") (str/ends-with? path "/api/v1/access_token")) - (let [form (read-form exchange) - grant (or (:grant form) "") - code (or (:code form) "mock:t2_test:redditor")] - (if (= grant "client_credentials") - (send-json exchange 200 "{\"access_token\":\"app-token\",\"token_type\":\"bearer\",\"expires_in\":3600}") - (send-json exchange 200 (str "{\"access_token\":\"" code "\",\"token_type\":\"bearer\",\"expires_in\":3600}")))) + ;; Reddit token (client_credentials for import + authorization_code for login) + (and (= method "POST") (str/ends-with? path "/api/v1/access_token")) + (let [form (read-form exchange) + grant (or (:grant form) "") + code (or (:code form) "mock:t2_test:redditor")] + (if (= grant "client_credentials") + (send-json exchange 200 "{\"access_token\":\"app-token\",\"token_type\":\"bearer\",\"expires_in\":3600}") + (send-json exchange 200 (str "{\"access_token\":\"" code "\",\"token_type\":\"bearer\",\"expires_in\":3600}")))) - ;; GitHub user - (= path "/user") - (let [token (bearer-token exchange) - user (or (parse-token-user token) {:id "1002" :login "newbie" :numeric? true})] - (send-json exchange 200 - (str "{\"id\":" (:id user) ",\"login\":\"" (:login user) "\"}"))) + ;; GitHub user + (= path "/user") + (let [token (bearer-token exchange) + user (or (parse-token-user token) {:id "1002" :login "newbie" :numeric? true})] + (send-json exchange 200 + (str "{\"id\":" (:id user) ",\"login\":\"" (:login user) "\"}"))) - ;; Reddit /api/v1/me - (str/ends-with? path "/api/v1/me") - (let [token (bearer-token exchange) - user (or (parse-token-user token) {:id "t2_test" :login "redditor"})] - (send-json exchange 200 - (str "{\"id\":\"" (:id user) "\",\"name\":\"" (:login user) "\"}"))) + ;; Reddit /api/v1/me + (str/ends-with? path "/api/v1/me") + (let [token (bearer-token exchange) + user (or (parse-token-user token) {:id "t2_test" :login "redditor"})] + (send-json exchange 200 + (str "{\"id\":\"" (:id user) "\",\"name\":\"" (:login user) "\"}"))) - :else - (send-json exchange 404 "{\"error\":\"not found\"}")))))] + :else + (send-json exchange 404 "{\"error\":\"not found\"}"))) + (catch Throwable t + (binding [*out* *err*] + (println "mock-oauth handler error:" t)) + (try + (send-json exchange 500 "{\"error\":\"mock-oauth internal\"}") + (catch Throwable _))))))] (.createContext server "/" handler) (.setExecutor server nil) (.start server) diff --git a/test/support/mock_reddit.clj b/test/support/mock_reddit.clj index a630cf0938722193e9af88382d60e777ff371be4..faa27945b6394363914c853627a0bad1e819a5f9 100644 --- a/test/support/mock_reddit.clj +++ b/test/support/mock_reddit.clj @@ -12,7 +12,7 @@ (defn- query-param [query key] (when query (some (fn [pair] - (let [[k v] (str/split pair "=" 2)] + (let [[k v] (str/split pair #"=" 2)] (when (= k key) (URLDecoder/decode (or v "") "UTF-8")))) (str/split query #"&")))) @@ -34,11 +34,11 @@ (defn- send-redirect [^HttpExchange ex location] (.set (.getResponseHeaders ex) "Location" location) - (.sendResponseHeaders ex 302 -1) + (.sendResponseHeaders ex 302 0) (.close (.getResponseBody ex))) (defn- read-form [^HttpExchange ex] - (let [body (slurp (.getInputStream ex))] + (let [body (slurp (.getRequestBody ex))] {:code (query-param body "code") :grant (query-param body "grant_type")})) @@ -48,7 +48,7 @@ (str/replace #"^[Bb]earer " ""))) (defn- parse-token-user [token] - (when (str/starts-with? token "mock:") + (when (and token (str/starts-with? token "mock:")) (parse-mock-user (subs token 5)))) (defn start-mock-reddit @@ -72,7 +72,7 @@ user (parse-mock-user (query-param query "mock_user")) code (str "mock:" (:id user) ":" (:login user)) loc (str redirect-uri "?code=" (java.net.URLEncoder/encode code "UTF-8") - "&state=" (java.net.URLEncoder/encode state "UTF-8"))] + "&state=" (java.net.URLEncoder/encode (or state "") "UTF-8"))] (send-redirect exchange loc)) (and (= method "POST") (str/ends-with? path "/api/v1/access_token")) Side B — contributor: tommy-mor Side B — commit message: [7bb7145d] url stuff Side B — unified diff (full patch): diff --git a/AGENTS.md b/AGENTS.md index e60b9ba6012593361ef10e8fdd9439cd9932e09b..babb889d6fbfb1fa7176c9e6b7544ae17b61dd2e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -58,4 +58,4 @@ Use **tmux** for `cargo run --package sorter2-server` (dev server). Rebuild afte - First `cargo test` / `cargo build --release` is slow; Clojure smoke test always does a release build. - `legacy/` and `ideas/` are not part of the workspace build. -- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/`](server/src/url_rules/) (composable Rust, not a config DSL). After changing canonicalization rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`. +- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/graph.rs`](server/src/url_rules/graph.rs): a semantic graph (DFA on host + path, query params in `Context`) with a generic internet fallback for unknown sites. After changing rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`. diff --git a/server/src/url_rules/engine.rs b/server/src/url_rules/engine.rs deleted file mode 100644 index e29b6b48c08deb7bffe031b1e542b1e25a7bef15..0000000000000000000000000000000000000000 --- a/server/src/url_rules/engine.rs +++ /dev/null @@ -1,187 +0,0 @@ -//! Composable URL normalization primitives. - -use std::collections::HashMap; - -use url::Url; - -/// Mutable URL view used by rule combinators before serializing to a canonical string. -#[derive(Debug, Clone)] -pub struct ParsedUrl { - pub scheme: String, - pub host: String, - pub path_segments: Vec, - pub query: HashMap, - pub fragment: Option, -} - -impl ParsedUrl { - pub fn parse(raw: &str) -> Option { - let trimmed = raw.trim(); - if trimmed.is_empty() { - return None; - } - - let with_scheme = if trimmed.contains("://") { - trimmed.to_string() - } else if trimmed.starts_with("r/") || trimmed.starts_with("/r/") { - let rest = trimmed.trim_start_matches('/').trim_start_matches("r/"); - format!("https://reddit.com/r/{rest}") - } else if trimmed.contains('.') && !trimmed.starts_with('/') { - format!("https://{trimmed}") - } else { - trimmed.to_string() - }; - - let url = Url::parse(&with_scheme).ok()?; - let host = url.host_str()?.to_string(); - let path_segments: Vec = url - .path_segments() - .map(|segs| segs.filter(|s| !s.is_empty()).map(str::to_string).collect()) - .unwrap_or_default(); - - let mut query = HashMap::new(); - for (k, v) in url.query_pairs() { - query.insert(k.into_owned(), v.into_owned()); - } - - Some(Self { - scheme: url.scheme().to_string(), - path_segments, - query, - fragment: url.fragment().map(str::to_string), - host, - }) - } - - pub fn with_path_segments(&self, segments: &[String]) -> Self { - let mut u = self.clone(); - u.path_segments = segments.to_vec(); - u - } - - pub fn to_url(&self) -> Option { - let mut url = if self.path_segments.is_empty() { - Url::parse(&format!("{}://{}", self.scheme, self.host)).ok()? - } else { - let path = format!("/{}", self.path_segments.join("/")); - Url::parse(&format!("{}://{}{}", self.scheme, self.host, path)).ok()? - }; - if !self.query.is_empty() { - let mut pairs: Vec<_> = self.query.iter().collect(); - pairs.sort_by(|a, b| a.0.cmp(b.0)); - url.query_pairs_mut().clear(); - for (k, v) in pairs { - url.query_pairs_mut().append_pair(k, v); - } - } - if let Some(ref frag) = self.fragment { - url.set_fragment(Some(frag)); - } - Some(url) - } - - pub fn canonical_string(&self) -> Option { - let url = self.to_url()?; - let mut s = url.to_string(); - if self.path_segments.is_empty() { - s = s.trim_end_matches('/').to_string(); - } - Some(s) - } -} - -pub fn force_https(u: &mut ParsedUrl) { - if u.scheme == "http" { - u.scheme = "https".to_string(); - } -} - -pub fn drop_fragment(u: &mut ParsedUrl) { - u.fragment = None; -} - -pub fn strip_www(u: &mut ParsedUrl) { - if u.host.starts_with("www.") { - u.host = u.host[4..].to_string(); - } -} - -pub fn lowercase_host(u: &mut ParsedUrl) { - u.host = u.host.to_ascii_lowercase(); -} - -pub fn lowercase_path(u: &mut ParsedUrl) { - for seg in &mut u.path_segments { - *seg = seg.to_ascii_lowercase(); - } -} - -pub fn clear_query(u: &mut ParsedUrl) { - u.query.clear(); -} - -pub fn keep_only_query(u: &mut ParsedUrl, keys: &[&str]) { - u.query - .retain(|k, _| keys.iter().any(|want| want == &k.as_str())); -} - -pub fn strip_tracking_params(u: &mut ParsedUrl) { - u.query.retain(|k, _| { - let lower = k.to_ascii_lowercase(); - !(lower.starts_with("utm_") - || matches!( - lower.as_str(), - "fbclid" | "gclid" | "ref" | "ref_src" | "ref_source" | "mc_cid" | "mc_eid" - )) - }); -} - -pub fn truncate_after_segment(u: &mut ParsedUrl, name: &str, keep: usize) { - if let Some(i) = u.path_segments.iter().position(|s| s == name) { - let end = (i + 1 + keep).min(u.path_segments.len()); - u.path_segments.truncate(end); - } -} - -pub fn drop_listing_suffix(u: &mut ParsedUrl, suffixes: &[&str]) { - if u.path_segments.len() >= 3 && u.path_segments.first().map(String::as_str) == Some("r") { - if let Some(last) = u.path_segments.last() { - if suffixes.iter().any(|s| *s == last.as_str()) { - u.path_segments.pop(); - } - } - } -} - -pub fn normalize_reddit_host(u: &mut ParsedUrl) { - if matches!( - u.host.as_str(), - "old.reddit.com" | "new.reddit.com" | "www.reddit.com" - ) { - u.host = "reddit.com".to_string(); - } -} - -pub fn rewrite_youtu_be(u: &mut ParsedUrl) { - if u.host == "youtu.be" && u.path_segments.len() == 1 { - let id = u.path_segments[0].clone(); - u.host = "youtube.com".to_string(); - u.path_segments = vec!["watch".to_string()]; - u.query.insert("v".to_string(), id); - } -} - -pub fn rewrite_youtube_shorts(u: &mut ParsedUrl) { - if u.host == "youtube.com" && u.path_segments.first().map(String::as_str) == Some("shorts") { - if let Some(id) = u.path_segments.get(1).cloned() { - u.path_segments = vec!["watch".to_string()]; - u.query.insert("v".to_string(), id); - } - } -} - -pub fn normalize_youtube_host(u: &mut ParsedUrl) { - if matches!(u.host.as_str(), "m.youtube.com" | "www.youtube.com") { - u.host = "youtube.com".to_string(); - } -} diff --git a/server/src/url_rules/mod.rs b/server/src/url_rules/mod.rs index 03d53bd3e82d704a01ba3fd8dd02b7d31422c0de..9e1445346ce77a49dd6a7e7713bf9c57aef353cc 100644 --- a/server/src/url_rules/mod.rs +++ b/server/src/url_rules/mod.rs @@ -1,8 +1,12 @@ -//! URL canonicalization and hierarchy rules for [`crate::path_types::ItemId`]. +//! URL canonicalization and hierarchy via a semantic graph (DFA + generic fallback). -mod engine; +mod graph; +mod parse; mod registry; +#[cfg(test)] +mod registry_tests; + pub use registry::{ canonicalize_raw, looks_like_url, navigable_breadcrumbs, parent_url, resolve_id, CanonicalResult, }; diff --git a/server/src/url_rules/registry.rs b/server/src/url_rules/registry.rs index 14514e9af8385fb2b9b2f35eb9ee14d453d4b97c..8e6c012ea1fc74b864307bdacdf5a0f5db5259fc 100644 --- a/server/src/url_rules/registry.rs +++ b/server/src/url_rules/registry.rs @@ -1,12 +1,7 @@ -//! Per-domain canonicalization and hierarchy rules. +//! Public API: canonical identity and hierarchy via the URL graph. -use std::collections::HashSet; - -use super::engine::{ - clear_query, drop_fragment, drop_listing_suffix, force_https, keep_only_query, lowercase_host, - lowercase_path, normalize_reddit_host, normalize_youtube_host, rewrite_youtu_be, - rewrite_youtube_shorts, strip_tracking_params, strip_www, truncate_after_segment, ParsedUrl, -}; +use super::graph::graph; +use super::parse::UrlParts; /// Result of canonicalizing a raw URL string. #[derive(Debug, Clone, PartialEq, Eq)] @@ -16,71 +11,16 @@ pub struct CanonicalResult { pub alias_of: Option, } -fn apply_global(u: &mut ParsedUrl) { - force_https(u); - drop_fragment(u); - strip_www(u); - lowercase_host(u); - strip_tracking_params(u); -} - -fn normalize_reddit(u: &mut ParsedUrl) { - normalize_reddit_host(u); - lowercase_path(u); - truncate_after_segment(u, "comments", 1); - drop_listing_suffix(u, &["hot", "top", "new", "rising", "controversial"]); - clear_query(u); -} - -fn normalize_youtube(u: &mut ParsedUrl) { - rewrite_youtu_be(u); - normalize_youtube_host(u); - rewrite_youtube_shorts(u); - keep_only_query(u, &["v", "list"]); -} - -fn normalize_default(_u: &mut ParsedUrl) { - // Global rules only. -} - -fn domain_key(host: &str) -> &'static str { - if host == "reddit.com" || host.ends_with(".reddit.com") { - "reddit.com" - } else if host == "youtube.com" || host == "youtu.be" { - "youtube.com" - } else { - "default" - } -} - -fn normalize_for_host(u: &mut ParsedUrl) { - apply_global(u); - match domain_key(&u.host) { - "reddit.com" => normalize_reddit(u), - "youtube.com" => normalize_youtube(u), - _ => normalize_default(u), - } -} - -/// Structural path segments that must not become standalone tree nodes when more path follows. -fn structural_trailing(host: &str) -> &'static [&'static str] { - match domain_key(host) { - "reddit.com" => &["comments"], - _ => &[], - } -} - /// Canonicalize a raw URL. Returns `None` if the input is not URL-like. pub fn canonicalize_raw(raw: &str) -> Option { let trimmed = raw.trim(); if trimmed.is_empty() { return None; } - let mut u = ParsedUrl::parse(trimmed)?; - let input_snapshot = u.canonical_string()?; - normalize_for_host(&mut u); - let canonical = u.canonical_string()?; - let alias_of = if input_snapshot != canonical { + let parts = UrlParts::parse(trimmed)?; + let g = graph(); + let canonical = g.resolve_canonical(&parts)?; + let alias_of = if trimmed != canonical { Some(trimmed.to_string()) } else { None @@ -98,35 +38,14 @@ pub fn resolve_id(raw: &str) -> Option { /// Navigable ancestor URLs from domain root up to and including `canonical` (full URLs). pub fn navigable_breadcrumbs(canonical: &str) -> Vec { - let Some(u) = ParsedUrl::parse(canonical) else { - return vec![canonical.to_string()]; + let parts = match UrlParts::parse(canonical) { + Some(p) => p, + None => return vec![canonical.to_string()], }; - let structural: HashSet<&str> = structural_trailing(&u.host).iter().copied().collect(); - let n = u.path_segments.len(); - let mut out = Vec::new(); - - // Domain root (no path segments). - if let Some(base) = u.with_path_segments(&[]).canonical_string() { - out.push(base); - } - - for i in 0..n { - let segs: Vec = u.path_segments[..=i].to_vec(); - let is_last = i == n - 1; - let seg = u.path_segments[i].as_str(); - if structural.contains(seg) && !is_last { - continue; - } - if let Some(url) = u.with_path_segments(&segs).canonical_string() { - if out.last() != Some(&url) { - out.push(url); - } - } - } - out + graph().breadcrumbs(&parts) } -/// Immediate parent scope URL, or `None` for tree root / opaque single-segment ids. +/// Immediate parent scope URL, or `None` for tree root. pub fn parent_url(canonical: &str) -> Option { let crumbs = navigable_breadcrumbs(canonical); if crumbs.len() <= 1 { @@ -148,88 +67,3 @@ pub fn looks_like_url(raw: &str) -> bool { || t.starts_with("youtu.be/") } -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn reddit_post_drops_slug_and_normalizes_host() { - let r = canonicalize_raw( - "https://old.reddit.com/r/AmItheAsshole/comments/1trnvdl/aita_for_cancelling/", - ) - .unwrap(); - assert_eq!( - r.canonical, - "https://reddit.com/r/amitheasshole/comments/1trnvdl" - ); - } - - #[test] - fn reddit_strips_query_and_listing() { - assert_eq!( - canonicalize_raw("https://www.reddit.com/r/rust/?sort=top") - .unwrap() - .canonical, - "https://reddit.com/r/rust" - ); - assert_eq!( - canonicalize_raw("https://www.reddit.com/r/programming/hot") - .unwrap() - .canonical, - "https://reddit.com/r/programming" - ); - } - - #[test] - fn reddit_short_path() { - assert_eq!( - canonicalize_raw("r/rust").unwrap().canonical, - "https://reddit.com/r/rust" - ); - } - - #[test] - fn reddit_breadcrumbs_skip_phantom_comments() { - let post = "https://reddit.com/r/aww/comments/1trnvdl"; - let crumbs = navigable_breadcrumbs(post); - assert!(!crumbs.iter().any(|c| c.ends_with("/comments"))); - assert_eq!( - crumbs.last().map(String::as_str), - Some(post) - ); - assert!(crumbs.contains(&"https://reddit.com/r/aww".to_string())); - } - - #[test] - fn reddit_parent_of_post_is_subreddit() { - assert_eq!( - parent_url("https://reddit.com/r/aww/comments/1trnvdl").as_deref(), - Some("https://reddit.com/r/aww") - ); - } - - #[test] - fn youtube_youtu_be_and_watch_same_canonical() { - let a = canonicalize_raw("https://youtu.be/dQw4w9WgXcQ").unwrap().canonical; - let b = canonicalize_raw("https://www.youtube.com/watch?v=dQw4w9WgXcQ&t=10").unwrap(); - assert_eq!(a, b.canonical); - assert_eq!(a, "https://youtube.com/watch?v=dQw4w9WgXcQ"); - } - - #[test] - fn legacy_schemeless_upgrades() { - assert_eq!( - canonicalize_raw("reddit.com/r/rust/comments/aaa/announcing_rust_199") - .unwrap() - .canonical, - "https://reddit.com/r/rust/comments/aaa" - ); - } - - #[test] - fn alias_recorded_when_input_differs() { - let r = canonicalize_raw("https://youtu.be/abc123").unwrap(); - assert_eq!(r.canonical, "https://youtube.com/watch?v=abc123"); - assert!(r.alias_of.is_some()); - } -}