You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [c59951f5] fixed canonical item paths business Side A — unified diff (full patch): diff --git a/server/src/html/breadcrumb_path.rs b/server/src/html/breadcrumb_path.rs index 12ad2c84faf2fbb693015d4552e45b5c54d587b9..c8a3937923161a6ff248bd77e87eff0dc6fe9ab0 100644 --- a/server/src/html/breadcrumb_path.rs +++ b/server/src/html/breadcrumb_path.rs @@ -1,4 +1,4 @@ -use crate::path_types::CanonicalItemUrl; +use crate::path_types::{tilde_http_path_to_canonical, CanonicalItemUrl}; /// Semantic view of an ontology path for rendering and routing decisions. pub(super) struct OntologyPath { @@ -11,16 +11,7 @@ impl OntologyPath { /// Path is the `*path` segment from `/~/*path` (e.g. `topic/a`). Always treat it as under `~/` /// so it canonicalizes to `https://slug.social/~/…`, not the non-tilde site path. pub(super) fn from_input(path: &str) -> Self { - let p = path.trim_start_matches('/'); - let raw = if p.starts_with("http://") || p.starts_with("https://") { - p.to_string() - } else if p.is_empty() { - "~/".to_string() - } else { - format!("~/{}", p) - }; - let canonical = CanonicalItemUrl::parse(&raw) - .unwrap_or_else(|| CanonicalItemUrl::parse("~/").unwrap()); + let canonical = tilde_http_path_to_canonical(path); Self::from_canonical(canonical) } @@ -37,7 +28,7 @@ impl OntologyPath { } pub(super) fn root() -> Self { - Self::from_canonical(CanonicalItemUrl::parse("~/").unwrap()) + Self::from_canonical(CanonicalItemUrl::ontology_root()) } pub(super) fn is_root(&self) -> bool { diff --git a/server/src/html/garden.rs b/server/src/html/garden.rs index d58d9b46f575eb6b7e4971086b07dda75ca2f645..319feb15a6b68d2b5df98b4289fedbc9bdd048d3 100644 --- a/server/src/html/garden.rs +++ b/server/src/html/garden.rs @@ -459,6 +459,8 @@ struct ItemPageViewModel { item: String, body: Option, sibling_rank: Option, + /// False at the tilde ontology root (`~/`): sibling-rank footnote does not apply. + item_has_parent: bool, child_rankings: ChildrenRankings, rank_history: Vec, /// Forum threads that mention or vote on this item. @@ -470,11 +472,12 @@ fn build_sibling_rank( scope: &ScopeId, item: &CanonicalItemUrl, ) -> Option { + let item = item.clone().normalized_storage(); let content = reduced .content_for_scope(scope) .unwrap_or_else(|| reduced.public()); let group = &content.ranking_group; - let parent = item.parent()?; + let parent = item.parent()?.normalized_storage(); let siblings: Vec = content .item_children .get(&parent) @@ -492,7 +495,7 @@ fn build_sibling_rank( if scoped_idxs.is_empty() { return None; } - let current_idx = *group.item_to_idx.get(item)?; + let current_idx = *group.item_to_idx.get(&item)?; if !scoped_idxs.contains(¤t_idx) { return None; } @@ -520,7 +523,7 @@ fn build_sibling_rank( .filter_map(|li| local_to_global.get(*li).copied()) .collect(); let ranked = ranked_items_subset(group, &comp_global, 10000, 1e-8); - let position = ranked.iter().position(|r| &r.item == item)? + 1; + let position = ranked.iter().position(|r| r.item == item)? + 1; Some(SiblingRank { position, component_size: ranked.len(), @@ -597,7 +600,9 @@ fn build_item_page_view_model( .content_for_scope(scope) .unwrap_or_else(|| reduced.public()); let item_key = CanonicalItemUrl::parse(item) - .unwrap_or_else(|| CanonicalItemUrl::parse("~/").unwrap()); + .unwrap_or_else(|| CanonicalItemUrl::parse("~/").unwrap()) + .normalized_storage(); + let item_has_parent = item_key.parent().is_some(); let child_rankings = build_children_rankings(content, &item_key); let rank_history = build_rank_history(reduced, scope, item_key.as_str()); @@ -617,6 +622,7 @@ fn build_item_page_view_model( .cloned() .or_else(|| reduced.public().item_bodies.get(&item_key).cloned()), sibling_rank: build_sibling_rank(reduced, scope, &item_key), + item_has_parent, child_rankings, rank_history, threads, @@ -652,7 +658,7 @@ async fn render_scope_view( (format!("#{} of {}", rank.position, rank.component_size)) } span class="muted" { (format!("({} siblings)", rank.sibling_total)) } - } @else { + } @else if model.item_has_parent { span class="muted" { "unranked among siblings" } } } @@ -815,6 +821,18 @@ mod tests { })); } + fn apply_ingest_room(state: &mut ReducerState, ts: i64, room_id: &str, raw: &str) { + state.apply_event(Event::Ingest(Ingest { + ts, + id: format!("ing-{ts}"), + raw: raw.to_string(), + principal: "testuser".to_string(), + delegate: Some("00000000-0000-0000-0000-000000000000:test:local/test".to_string()), + room_id: room_id.to_string(), + thread_tag: String::new(), + })); + } + #[test] fn item_page_model_includes_body_and_unranked_without_votes() { let mut reduced = ReducerState::default(); @@ -885,4 +903,85 @@ mod tests { || model.child_rankings.unranked_items.contains(&CanonicalItemUrl("https://slug.social/~/topic/kid2".to_string())) ); } + + #[test] + fn item_page_room_scope_root_lists_top_level_children() { + let mut reduced = ReducerState::default(); + apply_ingest_room( + &mut reduced, + 1, + "9ab12cd/my-room", + "@00000000-0000-0000-0000-000000000000:test:local/test\n~/t1 {a}\n~/t2 {b}\n", + ); + use crate::path_types::CanonicalItemUrl; + let root = CanonicalItemUrl::ontology_root(); + let model = build_item_page_view_model( + &reduced, + &ScopeId::Room("9ab12cd/my-room".to_string()), + root.as_str(), + ); + assert!(!model.item_has_parent); + assert_eq!(model.child_rankings.unranked_items.len(), 2); + let set: std::collections::HashSet<&str> = model + .child_rankings + .unranked_items + .iter() + .map(|u| u.as_str()) + .collect(); + assert!(set.contains("https://slug.social/~/t1")); + assert!(set.contains("https://slug.social/~/t2")); + } + + /// Top-level `~/a` vs `~/b` votes form one ranked component under the ontology root. + #[test] + fn item_page_room_scope_root_shows_ranked_child_group() { + let mut reduced = ReducerState::default(); + apply_ingest_room( + &mut reduced, + 1, + "9ab12cd/my-room", + "@00000000-0000-0000-0000-000000000000:test:local/test\n\ + ~/a {a}\n~/b {b}\n~/a 2:1 ~/b {because}\n", + ); + use crate::path_types::CanonicalItemUrl; + let root = CanonicalItemUrl::ontology_root(); + let model = build_item_page_view_model( + &reduced, + &ScopeId::Room("9ab12cd/my-room".to_string()), + root.as_str(), + ); + assert_eq!(model.child_rankings.component_rankings.len(), 1); + assert_eq!(model.child_rankings.component_rankings[0].pairs, 1); + let names: Vec<&str> = model.child_rankings.component_rankings[0] + .ranked + .iter() + .map(|r| r.item.as_str()) + .collect(); + assert_eq!( + names, + vec!["https://slug.social/~/a", "https://slug.social/~/b"] + ); + assert!(model.child_rankings.unranked_items.is_empty()); + } + + /// Legacy `https://slug.social/~/` spelling still resolves children under the real root key. + #[test] + fn item_page_model_normalizes_legacy_tilde_root_storage_url() { + let mut reduced = ReducerState::default(); + apply_ingest( + &mut reduced, + 1, + "@00000000-0000-0000-0000-000000000000:test:local/test\n~/x {x}\n", + ); + let model = build_item_page_view_model( + &reduced, + &ScopeId::Public, + "https://slug.social/~/", + ); + assert_eq!(model.child_rankings.unranked_items.len(), 1); + assert_eq!( + model.child_rankings.unranked_items[0].as_str(), + "https://slug.social/~/x" + ); + } } diff --git a/server/src/path_types.rs b/server/src/path_types.rs index 4c8075bbd488f1b9a5eda9e03ced83f6238c6d5e..361a9d446c9043cbdea5f060db2e8633c2fd9bf9 100644 --- a/server/src/path_types.rs +++ b/server/src/path_types.rs @@ -1,3 +1,5 @@ //! Re-exports — implementations live in `slug-types` (`paths` module). -pub use slug_types::paths::{CanonicalItemUrl, RelativePath, TildePath}; +pub use slug_types::paths::{ + tilde_http_path_to_canonical, CanonicalItemUrl, RelativePath, TildeHttpPathTail, TildePath, +}; diff --git a/server/src/scope_rank.rs b/server/src/scope_rank.rs index dc640848232b7e79142bfeeea3003c9023f84854..d656b2f0a0a3623ca6b234b746beaaca8ae6017d 100644 --- a/server/src/scope_rank.rs +++ b/server/src/scope_rank.rs @@ -149,9 +149,10 @@ pub fn build_rankings_for_item_set(content: &ContentState, items_in_scope: &[Can /// Build connected-component rankings for direct children of parent_scope. /// Matches the HTML garden view: multiple components, isolates, no-vote items. pub fn build_children_rankings(content: &ContentState, parent: &CanonicalItemUrl) -> ChildrenRankings { + let parent = parent.clone().normalized_storage(); let items: Vec = content .item_children - .get(parent) + .get(&parent) .map(|s| s.iter().cloned().collect()) .unwrap_or_default(); build_rankings_for_item_set(content, &items) diff --git a/server/tests/integration.rs b/server/tests/integration.rs index f9c218378f6a173e56ae1cec797b3c492986eac0..d4c5bfe9c6f1c71dc61878bd8c5e729b1b7c69ef 100644 --- a/server/tests/integration.rs +++ b/server/tests/integration.rs @@ -704,6 +704,62 @@ async fn test_private_room_post_links_use_private_garden_routes() { assert!(garden_body.contains(&format!("/r/{room_short}/{room_slug}/t/garden-thread"))); } +#[tokio::test] +async fn test_private_room_garden_root_lists_top_level_tilde_children() { + let (addr, _tmp, _log, _handle) = create_test_server().await; + let client = reqwest::Client::builder() + .redirect(reqwest::redirect::Policy::none()) + .build() + .unwrap(); + let bearer = test_bearer(); + + let create = rpc_batch( + &client, + addr, + Some(&bearer), + serde_json::json!([{ + "RoomCreate": { "slug": "garden-root-list" } + }]), + ) + .await; + let room_id = create["results"][0]["result"]["RoomCreated"]["room_id"] + .as_str() + .unwrap() + .to_string(); + let (room_short, room_slug) = room_id.split_once('/').unwrap(); + + let rpc = ui_post_ingest_rpc( + &room_id, + "ing", + "~/test1 {wow}\n~/test2 {wow2}\n~/test1 2:1 ~/test2 {because}\n", + ); + let post = client + .post(format!("http://{addr}/ui")) + .header("Authorization", format!("Bearer {bearer}")) + .form(&[("__rpc__", rpc.as_str())]) + .send() + .await + .unwrap(); + assert_eq!(post.status(), reqwest::StatusCode::OK); + + let root_page = client + .get(format!("http://{addr}/r/{room_short}/{room_slug}/~")) + .header("Authorization", format!("Bearer {bearer}")) + .send() + .await + .unwrap(); + assert!(root_page.status().is_success()); + let body = root_page.text().await.unwrap(); + assert!( + body.contains("ranked child groups"), + "expected garden child panel: {}", + body.len() + ); + assert!(body.contains("~/test1")); + assert!(body.contains("~/test2")); + assert!(body.contains("ordering 1")); +} + #[tokio::test] async fn test_post_check_returns_targeted_js_error_for_missing_thread_tag() { let (addr, _tmp, _log, _handle) = create_test_server().await; diff --git a/test/browser_public_garden.clj b/test/browser_public_garden.clj new file mode 100644 index 0000000000000000000000000000000000000000..7bea08e1cf2c8f5b1a94b0640a46e9ba5607b19b --- /dev/null +++ b/test/browser_public_garden.clj @@ -0,0 +1,100 @@ +(ns test.browser-public-garden + "Regression: ingest ontology items in the public room via RPC, then open /~ in the browser + and assert ranked + unranked paths appear (HTML garden index — not JSON API)." + (:require [babashka.fs :as fs] + [cheshire.core :as json] + [clojure.string :as str] + [clojure.test :refer [deftest is]] + [com.blockether.spel.core :as core] + [com.blockether.spel.locator :as locator] + [com.blockether.spel.page :as page] + [test.common :as common] + [test.oauth :as oauth])) + +(defn- wait-for-text [pg selector expected timeout-ms] + (let [deadline (+ (System/currentTimeMillis) timeout-ms)] + (loop [] + (let [text (locator/text-content (page/locator pg selector))] + (if (and (string? text) (str/includes? text expected)) + true + (if (< (System/currentTimeMillis) deadline) + (do (Thread/sleep 200) (recur)) + false)))))) + +(defn public-garden-flow! [] + (println "\n━━━ browser public room garden index (RPC seed + GET /~) ━━━\n") + + (common/letlocals + (bind build (common/run-cargo-build-release! ["slugsocial-server"])) + (is (zero? (:exit build)) "cargo build succeeds") + (bind server-bin "target/release/slugsocial-server") + + (bind tmp-dir (str (fs/create-temp-dir {:prefix "slug-browser-pub-garden-"}))) + (bind slug-port (common/pick-port)) + (bind google-port (common/pick-port)) + (bind base-url (str "http://127.0.0.1:" slug-port)) + (bind google-url (str "http://127.0.0.1:" google-port)) + + (bind !server (atom nil)) + (bind !google (atom nil)) + (bind server-env (common/slug-server-env tmp-dir base-url google-url slug-port)) + (try + (reset! !google (oauth/start-mock-google google-port + :google-users ["google-user-alice"])) + (reset! !server (common/start-server server-bin server-env)) + (is (common/wait-for-server base-url 10000) "server responds to /healthz") + + (let [alice-token (oauth/fetch-bearer-token! base-url :username "alice") + thread-tag "browser-pub-garden" + ;; Ranked pair at root + one isolate so both panels are exercised. + raw (str "# " thread-tag "\n\n" + "~/br-pub-a {alpha}\n" + "~/br-pub-b {beta}\n" + "~/br-pub-c {gamma}\n" + "~/br-pub-a 2:1 ~/br-pub-b {browser regression vote}\n") + post-resp (oauth/http-post-json + (str base-url "/api/v0/rpc") + [{"Post" {"room" "public" + "thread_tag" thread-tag + "text" raw + "return_rank_diff" false}}] + :headers {"Authorization" (str "Bearer " alice-token)}) + post-json (json/parse-string (:body post-resp) false) + _ (is (true? (get-in post-json ["results" 0 "ok"])) "seed public post via rpc") + rank-resp (oauth/http-post-json + (str base-url "/api/v0/rpc") + [{"GetGardenRank" {"room" "public" + "parent_path" "~" + "depth" 1}}]) + rank-json (json/parse-string (:body rank-resp) false) + _ (is (true? (get-in rank-json ["results" 0 "ok"])) "GetGardenRank ok") + comps (get-in rank-json ["results" 0 "result" "GardenRank" "components"]) + _ (is (pos? (count comps)) "rank API reports at least one component")] + (core/with-playwright [pw] + (core/with-browser [browser (core/launch-chromium pw {:headless true :channel "chrome"})] + (core/with-context [ctx (core/new-context browser)] + (core/with-page [pg (core/new-page-from-context ctx)] + (page/navigate pg (str base-url "/login")) + (is (wait-for-text pg "body" "@alice" 15000) "alice session after login") + (page/navigate pg (str base-url "/~")) + (is (wait-for-text pg "body" "paths" 15000) "public garden index shows paths heading") + (is (wait-for-text pg "body" "ordering" 10000) + "ranked child group meta visible") + (is (wait-for-text pg "body" "~/br-pub-a" 10000) "ranked list shows ~/br-pub-a") + (is (wait-for-text pg "body" "~/br-pub-b" 10000) "ranked list shows ~/br-pub-b") + (is (wait-for-text pg "body" "unranked" 10000) "unranked section present") + (is (wait-for-text pg "body" "~/br-pub-c" 10000) + "unranked list shows ~/br-pub-c")))))) + + (finally + (when-some [s @!server] (common/kill-server s)) + (when-some [g @!google] ((:stop-fn g))) + (fs/delete-tree tmp-dir))) + + nil)) + +(defn public-garden-browser-test [& _args] + (public-garden-flow!)) + +(deftest browser-public-garden-index-check + (public-garden-flow!)) diff --git a/tests.edn b/tests.edn index 56fdfa4f29383ad305fd8068783e4efe2ed334a0..6e96432ba766461b4233c75dc48db2633483e5e0 100644 --- a/tests.edn +++ b/tests.edn @@ -16,7 +16,8 @@ :ns-patterns ["^test\\.browser-sse$" "^test\\.browser-ui-morph$" "^test\\.browser-post-redact$" - "^test\\.browser-room-delete$"] + "^test\\.browser-room-delete$" + "^test\\.browser-public-garden$"] :kaocha.filter/skip-meta [:skip] :parallel? false}] :plugins [:kaocha.plugin/junit-xml] diff --git a/types/src/lib.rs b/types/src/lib.rs index 8cea90238ef6207d65ecdb2903b8958fafb213d2..edbb923e4d7d41ad82dfc254c3bd697562383527 100644 --- a/types/src/lib.rs +++ b/types/src/lib.rs @@ -4,8 +4,9 @@ pub mod paths; pub mod timeago; pub use paths::{ - canonicalize_item, canonicalize_tag, item_parent_path, item_path_segments, CanonicalItemUrl, - ForumThreadUrl, GardenItemUrl, RelativePath, TildeOntologyPath, TildePath, + canonicalize_item, canonicalize_tag, item_parent_path, item_path_segments, normalize_slug_ontology_storage_url, + CanonicalItemUrl, ForumThreadUrl, GardenItemUrl, RelativePath, SLUG_TILDE_ONTOLOGY_ROOT, + TildeHttpPathTail, TildeOntologyPath, TildePath, tilde_http_path_to_canonical, }; /// Max characters returned for a garden item body unless `full=true` / `--full` (API + CLI). diff --git a/types/src/paths.rs b/types/src/paths.rs index 668323ed777f360f2838a0dfe6eae68905d1183b..8971f599a7ebcf399f60c306d35ee28d0f581194 100644 --- a/types/src/paths.rs +++ b/types/src/paths.rs @@ -1,5 +1,13 @@ //! Canonical paths, storage ids, and JSON href newtypes. All normalization and //! room-aware URL rules for items live here. +//! +//! ## String kinds (parse in this module only) +//! +//! - **[`canonicalize_item`] / [`CanonicalItemUrl`]** — graph storage key and DSL form; tilde +//! ontology root is always [`SLUG_TILDE_ONTOLOGY_ROOT`] (no `…/~/` trailing slash only). +//! - **[`TildeHttpPathTail`]** — capture from `GET /~/*path` or `…/r/…/~/…` (the `*path` segment). +//! - **`-/…` wire form** — external items; see [`canonicalize_item`] dash branch. +//! - **[`GardenItemUrl`], [`ForumThreadUrl`]** — JSON / browser href surfaces. use std::borrow::Borrow; use std::fmt; @@ -7,6 +15,23 @@ use std::ops::Deref; use serde::{Deserialize, Serialize}; +// --------------------------------------------------------------------------- +// Slug tilde ontology (single storage form for `~/`) +// --------------------------------------------------------------------------- + +/// Canonical absolute URL for the tilde ontology **root** (`~/` in UI). Used as the +/// `item_children` parent key for top-level items and must match [`CanonicalItemUrl::ontology_root`]. +pub const SLUG_TILDE_ONTOLOGY_ROOT: &str = "https://slug.social/~"; + +/// Collapse legacy or parser variants of the ontology root to [`SLUG_TILDE_ONTOLOGY_ROOT`]. +pub fn normalize_slug_ontology_storage_url(s: &str) -> String { + if s == "https://slug.social/~/" { + SLUG_TILDE_ONTOLOGY_ROOT.to_string() + } else { + s.to_string() + } +} + // --------------------------------------------------------------------------- // Normalization (moved from server `canonical_path`) // --------------------------------------------------------------------------- @@ -89,6 +114,9 @@ pub fn canonicalize_item(input: &str) -> String { .join("/"); if is_tilde { + if tail.is_empty() { + return SLUG_TILDE_ONTOLOGY_ROOT.to_string(); + } format!("https://slug.social/~/{}", tail) } else if tail.is_empty() { "https://slug.social".to_string() @@ -173,7 +201,7 @@ impl CanonicalItemUrl { if c.is_empty() { None } else { - Some(Self(c)) + Some(Self(normalize_slug_ontology_storage_url(&c))) } } @@ -181,8 +209,19 @@ impl CanonicalItemUrl { &self.0 } + /// Collapses legacy slug ontology root spellings so [`HashMap`] keys match the reducer graph. + pub fn normalized_storage(self) -> Self { + Self(normalize_slug_ontology_storage_url(self.as_str())) + } + pub fn tilde_tail(&self) -> Option<&str> { - self.0.strip_prefix("https://slug.social/~/") + if let Some(tail) = self.0.strip_prefix("https://slug.social/~/") { + return Some(tail); + } + if self.0 == SLUG_TILDE_ONTOLOGY_ROOT || self.0 == "https://slug.social/~/" { + return Some(""); + } + None } pub fn last_segment(&self) -> &str { @@ -193,7 +232,7 @@ impl CanonicalItemUrl { } pub fn ontology_root() -> Self { - Self("https://slug.social/~".to_string()) + Self(SLUG_TILDE_ONTOLOGY_ROOT.to_string()) } pub fn parent(&self) -> Option { @@ -234,9 +273,13 @@ impl CanonicalItemUrl { /// `-/` representation for external `https://…` items, `~/…` for slug ontology, else unchanged. pub fn display_path(&self) -> String { - if let Some(tail) = self.0.strip_prefix("https://slug.social/~/") { - format!("~/{}", tail) - } else if let Some(tail) = self.0.strip_prefix("https://") { + if let Some(tail) = self.tilde_tail() { + if tail.is_empty() { + return "~/".to_string(); + } + return format!("~/{}", tail); + } + if let Some(tail) = self.0.strip_prefix("https://") { if tail.starts_with("slug.social") { self.0.clone() } else { @@ -276,6 +319,37 @@ impl CanonicalItemUrl { } } +/// HTTP route capture: path segment after `~/` in `GET /~/*path` or `…/r/…/~/…` (empty = ontology root). +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct TildeHttpPathTail(pub String); + +impl TildeHttpPathTail { + pub fn new(path_segment: &str) -> Self { + Self(path_segment.trim_start_matches('/').to_string()) + } + + pub fn as_str(&self) -> &str { + &self.0 + } + + pub fn to_canonical(&self) -> CanonicalItemUrl { + tilde_http_path_to_canonical(self.as_str()) + } +} + +/// Map the router's tilde tail (e.g. `topic/a`, or empty for root) to a [`CanonicalItemUrl`]. +pub fn tilde_http_path_to_canonical(path_segment: &str) -> CanonicalItemUrl { + let p = path_segment.trim_start_matches('/'); + let raw = if p.starts_with("http://") || p.starts_with("https://") { + p.to_string() + } else if p.is_empty() { + "~/".to_string() + } else { + format!("~/{}", p) + }; + CanonicalItemUrl::parse(&raw).unwrap_or_else(|| CanonicalItemUrl::ontology_root()) +} + impl fmt::Display for CanonicalItemUrl { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { self.0.fmt(f) @@ -557,6 +631,44 @@ mod tests { fn canonical_parent_root_is_none() { let root = CanonicalItemUrl::parse("~/").unwrap(); assert!(root.parent().is_none()); + assert_eq!(root.as_str(), SLUG_TILDE_ONTOLOGY_ROOT); + assert_eq!(root, CanonicalItemUrl::ontology_root()); + } + + #[test] + fn tilde_ontology_root_unifies_forms() { + assert_eq!(canonicalize_item("~/"), SLUG_TILDE_ONTOLOGY_ROOT); + assert_eq!( + normalize_slug_ontology_storage_url("https://slug.social/~/"), + SLUG_TILDE_ONTOLOGY_ROOT.to_string() + ); + assert_eq!( + CanonicalItemUrl::parse("https://slug.social/~/") + .unwrap() + .as_str(), + SLUG_TILDE_ONTOLOGY_ROOT + ); + let legacy = CanonicalItemUrl("https://slug.social/~/".to_string()); + assert_eq!(legacy.normalized_storage().as_str(), SLUG_TILDE_ONTOLOGY_ROOT); + } + + #[test] + fn tilde_http_path_tail_maps_router_segment() { + assert_eq!( + TildeHttpPathTail::new("").to_canonical(), + CanonicalItemUrl::ontology_root() + ); + assert_eq!( + tilde_http_path_to_canonical("topic/x").as_str(), + "https://slug.social/~/topic/x" + ); + } + + #[test] + fn display_path_slug_ontology_root() { + let r = CanonicalItemUrl::ontology_root(); + assert_eq!(r.display_path(), "~/"); + assert_eq!(r.tilde_tail(), Some("")); } #[test] Side B — contributor: tommy-mor Side B — commit message: [8d8230d1] reddit Side B — unified diff (full patch): diff --git a/.gitignore b/.gitignore index 4c7073f9fac0c30fd2050d79a60ef447af58ebeb..ada462e900d24a3a6d08165d158c80f79c35a5a5 100644 --- a/.gitignore +++ b/.gitignore @@ -7,3 +7,4 @@ data/ repomix-output.xml dev-data/ +.env diff --git a/server/Cargo.toml b/server/Cargo.toml index 7906a8547d56b8e6a48ef59c37aa82a8510fdee9..4677fedcb45292eebebe7e9cf6ce2f5738f18ddf 100644 --- a/server/Cargo.toml +++ b/server/Cargo.toml @@ -16,6 +16,7 @@ tower = "0.5" tower-http = { version = "0.5", features = ["trace"] } tracing = "0.1" tracing-subscriber = { version = "0.3", features = ["env-filter"] } +reqwest = { version = "0.12", features = ["json"] } [dev-dependencies] reqwest = { version = "0.12", features = ["json"] } diff --git a/server/src/html/mod.rs b/server/src/html/mod.rs index 2864407ed6e8ec284a1dc663acf1805534566b62..df6505021d9f446c2b453e20e3eb3cf696a111f9 100644 --- a/server/src/html/mod.rs +++ b/server/src/html/mod.rs @@ -272,5 +272,16 @@ pub async fn home(State(state): State, uri: Uri) -> impl IntoResponse pub async fn browse(State(state): State, uri: Uri) -> impl IntoResponse { let item = ItemId::from_browse_uri(uri.path()).unwrap_or(ItemId::root()); + if item.as_str().starts_with("reddit.com") { + let needs_fetch = { + let tree = state.tree.read().await; + tree.get(&item) + .map(|n| n.data.is_none()) + .unwrap_or(true) + }; + if needs_fetch { + state.reddit.request_fetch(item.clone()); + } + } item_page(state, uri, item).await } diff --git a/server/src/reddit.rs b/server/src/reddit.rs index d203dca09245daf869b3aa942898447700ae69fb..90053ad03b1d7c8e94f325dd4ee64c2b4f7da900 100644 --- a/server/src/reddit.rs +++ b/server/src/reddit.rs @@ -1,4 +1,12 @@ -//! Reddit API import (async, decoupled from UI request path). +//! Reddit API import via a single background worker (rate limits, dedup, backoff). + +use std::collections::{HashMap, HashSet}; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use reqwest::{header, Client, StatusCode}; +use serde::Deserialize; +use tokio::sync::{mpsc, RwLock}; use crate::{ path_types::ItemId, @@ -10,12 +18,401 @@ pub fn ensure_partial_tree(tree: &mut GlobalTree, id: &ItemId) { tree.ensure_path(id); } -/// Placeholder for Reddit JSON import. Returns entity data when implemented. -pub async fn fetch_reddit_entity(_id: &ItemId) -> Option { - None +pub struct RedditCommand { + pub id: ItemId, +} + +#[derive(Clone)] +pub struct RedditBroker { + tx: mpsc::Sender, +} + +#[derive(Clone)] +struct RedditCredentials { + client_id: String, + client_secret: String, +} + +struct OAuthToken { + access_token: String, + expires_at: Instant, +} + +impl RedditBroker { + pub fn spawn(tree: Arc>, user_agent: &str) -> Self { + let (tx, rx) = mpsc::channel(100); + + let mut headers = header::HeaderMap::new(); + headers.insert( + header::USER_AGENT, + header::HeaderValue::from_str(user_agent).expect("valid user agent"), + ); + + let client = Client::builder() + .default_headers(headers) + .timeout(Duration::from_secs(15)) + .build() + .expect("reqwest client"); + + let creds = RedditCredentials::from_env(); + tokio::spawn(reddit_worker(rx, tree, client, creds)); + + Self { tx } + } + + /// Fire-and-forget: queue a fetch; worker updates the tree when done. + pub fn request_fetch(&self, id: ItemId) { + let _ = self.tx.try_send(RedditCommand { id }); + } +} + +impl RedditCredentials { + fn from_env() -> Option { + let client_id = std::env::var("REDDIT_CLIENT_ID").ok()?; + let client_secret = std::env::var("REDDIT_CLIENT_SECRET").ok()?; + if client_id.is_empty() || client_secret.is_empty() { + return None; + } + Some(Self { + client_id, + client_secret, + }) + } +} + +pub fn default_user_agent() -> String { + std::env::var("REDDIT_USER_AGENT").unwrap_or_else(|_| { + "web:sorter2.social:v0.0.1 (by /u/sorter2)".to_string() + }) } -/// Apply fetched entity data to a node (called from async worker). -pub fn apply_entity(tree: &mut GlobalTree, id: &ItemId, data: EntityData) { - tree.set_entity_data(id, data); +async fn reddit_worker( + mut rx: mpsc::Receiver, + tree: Arc>, + client: Client, + creds: Option, +) { + let mut in_flight = HashSet::new(); + let mut recently_fetched: HashMap = HashMap::new(); + let mut current_delay = Duration::from_secs(1); + let mut oauth: Option = None; + let cache_ttl = Duration::from_secs(300); + + while let Some(cmd) = rx.recv().await { + let now = Instant::now(); + recently_fetched.retain(|_, t| now.duration_since(*t) < cache_ttl); + + if in_flight.contains(&cmd.id) || recently_fetched.contains_key(&cmd.id) { + continue; + } + + in_flight.insert(cmd.id.clone()); + let fetch_id = cmd.id.clone(); + + tokio::time::sleep(current_delay).await; + + if let Some(c) = &creds { + oauth = ensure_oauth_token(&client, c, oauth.take()).await; + } + + let token = oauth.as_ref().map(|t| t.access_token.as_str()); + let use_oauth = token.is_some(); + + match do_fetch(&client, &fetch_id, use_oauth, token).await { + Ok(FetchOutcome::Entity(data)) => { + let mut w = tree.write().await; + w.set_entity_data(&fetch_id, data); + recently_fetched.insert(fetch_id.clone(), Instant::now()); + current_delay = Duration::from_millis(600); + } + Ok(FetchOutcome::NotFound) => { + recently_fetched.insert(fetch_id.clone(), Instant::now()); + } + Ok(FetchOutcome::RateLimited { reset_secs }) => { + let wait = Duration::from_secs(reset_secs.max(1)); + tracing::warn!( + "Reddit rate limit for {}; sleeping {}s", + fetch_id, + wait.as_secs() + ); + tokio::time::sleep(wait).await; + current_delay = (current_delay * 2).min(Duration::from_secs(60)); + } + Err(e) => { + tracing::warn!("Reddit fetch failed for {}: {}", fetch_id, e); + current_delay = (current_delay * 2).min(Duration::from_secs(60)); + } + } + + in_flight.remove(&fetch_id); + } +} + +enum FetchOutcome { + Entity(EntityData), + NotFound, + RateLimited { reset_secs: u64 }, +} + +async fn ensure_oauth_token( + client: &Client, + creds: &RedditCredentials, + existing: Option, +) -> Option { + if let Some(t) = existing { + if Instant::now() < t.expires_at - Duration::from_secs(60) { + return Some(t); + } + } + + let resp = client + .post("https://www.reddit.com/api/v1/access_token") + .basic_auth(&creds.client_id, Some(&creds.client_secret)) + .form(&[("grant_type", "client_credentials")]) + .send() + .await; + + let resp = match resp { + Ok(r) => r, + Err(e) => { + tracing::warn!("Reddit OAuth token request failed: {e}"); + return None; + } + }; + + if !resp.status().is_success() { + tracing::warn!("Reddit OAuth token HTTP {}", resp.status()); + return None; + } + + #[derive(Deserialize)] + struct TokenResponse { + access_token: String, + expires_in: u64, + } + + let body: TokenResponse = match resp.json().await { + Ok(b) => b, + Err(e) => { + tracing::warn!("Reddit OAuth token parse failed: {e}"); + return None; + } + }; + + Some(OAuthToken { + access_token: body.access_token, + expires_at: Instant::now() + Duration::from_secs(body.expires_in), + }) +} + +async fn do_fetch( + client: &Client, + id: &ItemId, + use_oauth: bool, + bearer: Option<&str>, +) -> Result { + let url = map_item_to_reddit_api(id, use_oauth); + if url.is_empty() { + return Ok(FetchOutcome::NotFound); + } + + let mut req = client.get(&url); + if let Some(token) = bearer { + req = req.bearer_auth(token); + } + + let resp = req.send().await.map_err(|e| e.to_string())?; + + if resp.status() == StatusCode::TOO_MANY_REQUESTS { + let reset = rate_limit_reset_secs(&resp); + return Ok(FetchOutcome::RateLimited { reset_secs: reset }); + } + + if resp.status() == StatusCode::SERVICE_UNAVAILABLE { + return Err("Reddit unavailable (503)".to_string()); + } + + if !resp.status().is_success() { + return Ok(FetchOutcome::NotFound); + } + + if rate_limit_remaining(&resp) == Some(0) { + let reset = rate_limit_reset_secs(&resp); + return Ok(FetchOutcome::RateLimited { reset_secs: reset }); + } + + let bytes = resp.bytes().await.map_err(|e| e.to_string())?; + Ok(parse_reddit_json(id, &bytes) + .map(FetchOutcome::Entity) + .unwrap_or(FetchOutcome::NotFound)) +} + +fn rate_limit_remaining(resp: &reqwest::Response) -> Option { + resp.headers() + .get("x-ratelimit-remaining") + .and_then(|v| v.to_str().ok()) + .and_then(|s| s.parse::().ok()) + .map(|f| f.floor() as u64) +} + +fn rate_limit_reset_secs(resp: &reqwest::Response) -> u64 { + resp.headers() + .get("x-ratelimit-reset") + .and_then(|v| v.to_str().ok()) + .and_then(|s| s.parse::().ok()) + .map(|f| f.ceil() as u64) + .unwrap_or(5) +} + +/// Map canonical item id to Reddit JSON API URL. +pub fn map_item_to_reddit_api(id: &ItemId, oauth: bool) -> String { + let path = id.as_str(); + if !path.starts_with("reddit.com/") && path != "reddit.com" { + return String::new(); + } + + let base = if oauth { + "https://oauth.reddit.com" + } else { + "https://www.reddit.com" + }; + + let segments: Vec<&str> = path.split('/').collect(); + + if let Some(i) = segments.iter().position(|&p| p == "comments") { + if segments.len() > i + 1 { + let api_path = segments[1..=i + 1].join("/"); + return format!("{base}/{api_path}.json?raw_json=1"); + } + } + + if segments.len() == 3 && segments[1] == "r" { + return format!("{base}/r/{}/about.json?raw_json=1", segments[2]); + } + + String::new() +} + +fn parse_reddit_json(id: &ItemId, bytes: &[u8]) -> Option { + let v: serde_json::Value = serde_json::from_slice(bytes).ok()?; + let segments: Vec<&str> = id.as_str().split('/').collect(); + + if segments.iter().any(|&p| p == "comments") { + parse_post_listing(&v) + } else { + parse_subreddit_about(&v) + } +} + +fn parse_subreddit_about(v: &serde_json::Value) -> Option { + let data = v.get("data")?; + let title = data + .get("title") + .or_else(|| data.get("display_name")) + .and_then(|t| t.as_str())? + .to_string(); + let body_html = data + .get("public_description_html") + .or_else(|| data.get("public_description")) + .and_then(|t| t.as_str()) + .map(|s| s.to_string()); + let thumb_url = data + .get("icon_img") + .or_else(|| data.get("community_icon")) + .and_then(|t| t.as_str()) + .filter(|s| !s.is_empty()) + .map(|s| s.to_string()); + + Some(EntityData { + title, + author: None, + body_html, + thumb_url, + }) +} + +fn parse_post_listing(v: &serde_json::Value) -> Option { + let listing = v.as_array()?.first()?; + let child = listing + .pointer("/data/children/0/data")?; + let title = child.get("title")?.as_str()?.to_string(); + let author = child + .get("author") + .and_then(|a| a.as_str()) + .filter(|a| *a != "[deleted]") + .map(|s| s.to_string()); + let body_html = child + .get("selftext_html") + .and_then(|t| t.as_str()) + .filter(|s| !s.is_empty()) + .map(|s| s.to_string()); + let thumb_url = child + .get("thumbnail") + .and_then(|t| t.as_str()) + .filter(|s| s.starts_with("http")) + .map(|s| s.to_string()); + + Some(EntityData { + title, + author, + body_html, + thumb_url, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn map_subreddit_about_url() { + let id = ItemId::parse("reddit.com/r/rust").unwrap(); + assert_eq!( + map_item_to_reddit_api(&id, false), + "https://www.reddit.com/r/rust/about.json?raw_json=1" + ); + assert_eq!( + map_item_to_reddit_api(&id, true), + "https://oauth.reddit.com/r/rust/about.json?raw_json=1" + ); + } + + #[test] + fn map_post_url() { + let id = + ItemId::parse("reddit.com/r/amitheasshole/comments/1trnvdl").unwrap(); + assert_eq!( + map_item_to_reddit_api(&id, false), + "https://www.reddit.com/r/amitheasshole/comments/1trnvdl.json?raw_json=1" + ); + } + + #[test] + fn map_non_reddit_empty() { + let id = ItemId::opaque("example.com/foo"); + assert!(map_item_to_reddit_api(&id, false).is_empty()); + } + + #[test] + fn parse_subreddit_fixture() { + let json = r#"{"kind":"t5","data":{"title":"Rust","display_name":"rust","public_description":"systems"}}"#; + let entity = parse_reddit_json( + &ItemId::parse("reddit.com/r/rust").unwrap(), + json.as_bytes(), + ) + .unwrap(); + assert_eq!(entity.title, "Rust"); + } + + #[test] + fn parse_post_fixture() { + let json = r#"[{"kind":"Listing","data":{"children":[{"kind":"t3","data":{"title":"AITA","author":"op","selftext_html":"<p>hi</p>","thumbnail":"https://b.thumbs.redditmedia.com/x.jpg"}}]}}]"#; + let entity = parse_reddit_json( + &ItemId::parse("reddit.com/r/x/comments/abc").unwrap(), + json.as_bytes(), + ) + .unwrap(); + assert_eq!(entity.title, "AITA"); + assert_eq!(entity.author.as_deref(), Some("op")); + } } diff --git a/server/src/state.rs b/server/src/state.rs index cc1722f5a5bf4d415f2327ea585c488a15a75592..8c03aa60c15aee803a534439400b69935b1a3d84 100644 --- a/server/src/state.rs +++ b/server/src/state.rs @@ -5,9 +5,10 @@ use tokio::sync::RwLock; use crate::{ event_log::EventLog, events::Event, + journal::JournalClient, path_types::ItemId, + reddit::{default_user_agent, RedditBroker}, reducer::{GlobalTree, VoteData}, - journal::JournalClient, views::ViewStore, }; @@ -73,6 +74,7 @@ pub struct AppState { pub views: ViewStore, pub tree: Arc>, journal: JournalClient, + pub reddit: RedditBroker, } impl AppState { @@ -112,6 +114,7 @@ impl AppState { let tree = Arc::new(RwLock::new(tree)); let journal = JournalClient::spawn(tree.clone(), event_log.clone()); + let reddit = RedditBroker::spawn(tree.clone(), &default_user_agent()); Self { cfg: Arc::new(cfg), @@ -119,6 +122,7 @@ impl AppState { views, tree, journal, + reddit, } } @@ -127,8 +131,11 @@ impl AppState { id: id.as_str().to_string(), }; self.event_log.append(&event).await.map_err(|e| e.to_string())?; - let mut w = self.tree.write().await; - w.ensure_path(id); + { + let mut w = self.tree.write().await; + w.ensure_path(id); + } + self.reddit.request_fetch(id.clone()); Ok(()) } diff --git a/todo b/todo new file mode 100644 index 0000000000000000000000000000000000000000..d196e8cb4cc80ccb95eeff01c73607d520e13212 --- /dev/null +++ b/todo @@ -0,0 +1,7 @@ +reddit import (only on explicit request) +reddit rendering +vote redering +pair chosing +nsfw gate + +logins (uuid user, two sides, oauths, and pseudonyms)