You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [7bb7145d] url stuff Side A — unified diff (full patch): diff --git a/AGENTS.md b/AGENTS.md index e60b9ba6012593361ef10e8fdd9439cd9932e09b..babb889d6fbfb1fa7176c9e6b7544ae17b61dd2e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -58,4 +58,4 @@ Use **tmux** for `cargo run --package sorter2-server` (dev server). Rebuild afte - First `cargo test` / `cargo build --release` is slow; Clojure smoke test always does a release build. - `legacy/` and `ideas/` are not part of the workspace build. -- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/`](server/src/url_rules/) (composable Rust, not a config DSL). After changing canonicalization rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`. +- **ItemId** for web URLs is a canonical full URL (`https://reddit.com/r/rust`). Rules live in [`server/src/url_rules/graph.rs`](server/src/url_rules/graph.rs): a semantic graph (DFA on host + path, query params in `Context`) with a generic internet fallback for unknown sites. After changing rules, rebuild the projection: `cargo run --package sorter2-server -- replay-index`. diff --git a/server/src/url_rules/engine.rs b/server/src/url_rules/engine.rs deleted file mode 100644 index e29b6b48c08deb7bffe031b1e542b1e25a7bef15..0000000000000000000000000000000000000000 --- a/server/src/url_rules/engine.rs +++ /dev/null @@ -1,187 +0,0 @@ -//! Composable URL normalization primitives. - -use std::collections::HashMap; - -use url::Url; - -/// Mutable URL view used by rule combinators before serializing to a canonical string. -#[derive(Debug, Clone)] -pub struct ParsedUrl { - pub scheme: String, - pub host: String, - pub path_segments: Vec, - pub query: HashMap, - pub fragment: Option, -} - -impl ParsedUrl { - pub fn parse(raw: &str) -> Option { - let trimmed = raw.trim(); - if trimmed.is_empty() { - return None; - } - - let with_scheme = if trimmed.contains("://") { - trimmed.to_string() - } else if trimmed.starts_with("r/") || trimmed.starts_with("/r/") { - let rest = trimmed.trim_start_matches('/').trim_start_matches("r/"); - format!("https://reddit.com/r/{rest}") - } else if trimmed.contains('.') && !trimmed.starts_with('/') { - format!("https://{trimmed}") - } else { - trimmed.to_string() - }; - - let url = Url::parse(&with_scheme).ok()?; - let host = url.host_str()?.to_string(); - let path_segments: Vec = url - .path_segments() - .map(|segs| segs.filter(|s| !s.is_empty()).map(str::to_string).collect()) - .unwrap_or_default(); - - let mut query = HashMap::new(); - for (k, v) in url.query_pairs() { - query.insert(k.into_owned(), v.into_owned()); - } - - Some(Self { - scheme: url.scheme().to_string(), - path_segments, - query, - fragment: url.fragment().map(str::to_string), - host, - }) - } - - pub fn with_path_segments(&self, segments: &[String]) -> Self { - let mut u = self.clone(); - u.path_segments = segments.to_vec(); - u - } - - pub fn to_url(&self) -> Option { - let mut url = if self.path_segments.is_empty() { - Url::parse(&format!("{}://{}", self.scheme, self.host)).ok()? - } else { - let path = format!("/{}", self.path_segments.join("/")); - Url::parse(&format!("{}://{}{}", self.scheme, self.host, path)).ok()? - }; - if !self.query.is_empty() { - let mut pairs: Vec<_> = self.query.iter().collect(); - pairs.sort_by(|a, b| a.0.cmp(b.0)); - url.query_pairs_mut().clear(); - for (k, v) in pairs { - url.query_pairs_mut().append_pair(k, v); - } - } - if let Some(ref frag) = self.fragment { - url.set_fragment(Some(frag)); - } - Some(url) - } - - pub fn canonical_string(&self) -> Option { - let url = self.to_url()?; - let mut s = url.to_string(); - if self.path_segments.is_empty() { - s = s.trim_end_matches('/').to_string(); - } - Some(s) - } -} - -pub fn force_https(u: &mut ParsedUrl) { - if u.scheme == "http" { - u.scheme = "https".to_string(); - } -} - -pub fn drop_fragment(u: &mut ParsedUrl) { - u.fragment = None; -} - -pub fn strip_www(u: &mut ParsedUrl) { - if u.host.starts_with("www.") { - u.host = u.host[4..].to_string(); - } -} - -pub fn lowercase_host(u: &mut ParsedUrl) { - u.host = u.host.to_ascii_lowercase(); -} - -pub fn lowercase_path(u: &mut ParsedUrl) { - for seg in &mut u.path_segments { - *seg = seg.to_ascii_lowercase(); - } -} - -pub fn clear_query(u: &mut ParsedUrl) { - u.query.clear(); -} - -pub fn keep_only_query(u: &mut ParsedUrl, keys: &[&str]) { - u.query - .retain(|k, _| keys.iter().any(|want| want == &k.as_str())); -} - -pub fn strip_tracking_params(u: &mut ParsedUrl) { - u.query.retain(|k, _| { - let lower = k.to_ascii_lowercase(); - !(lower.starts_with("utm_") - || matches!( - lower.as_str(), - "fbclid" | "gclid" | "ref" | "ref_src" | "ref_source" | "mc_cid" | "mc_eid" - )) - }); -} - -pub fn truncate_after_segment(u: &mut ParsedUrl, name: &str, keep: usize) { - if let Some(i) = u.path_segments.iter().position(|s| s == name) { - let end = (i + 1 + keep).min(u.path_segments.len()); - u.path_segments.truncate(end); - } -} - -pub fn drop_listing_suffix(u: &mut ParsedUrl, suffixes: &[&str]) { - if u.path_segments.len() >= 3 && u.path_segments.first().map(String::as_str) == Some("r") { - if let Some(last) = u.path_segments.last() { - if suffixes.iter().any(|s| *s == last.as_str()) { - u.path_segments.pop(); - } - } - } -} - -pub fn normalize_reddit_host(u: &mut ParsedUrl) { - if matches!( - u.host.as_str(), - "old.reddit.com" | "new.reddit.com" | "www.reddit.com" - ) { - u.host = "reddit.com".to_string(); - } -} - -pub fn rewrite_youtu_be(u: &mut ParsedUrl) { - if u.host == "youtu.be" && u.path_segments.len() == 1 { - let id = u.path_segments[0].clone(); - u.host = "youtube.com".to_string(); - u.path_segments = vec!["watch".to_string()]; - u.query.insert("v".to_string(), id); - } -} - -pub fn rewrite_youtube_shorts(u: &mut ParsedUrl) { - if u.host == "youtube.com" && u.path_segments.first().map(String::as_str) == Some("shorts") { - if let Some(id) = u.path_segments.get(1).cloned() { - u.path_segments = vec!["watch".to_string()]; - u.query.insert("v".to_string(), id); - } - } -} - -pub fn normalize_youtube_host(u: &mut ParsedUrl) { - if matches!(u.host.as_str(), "m.youtube.com" | "www.youtube.com") { - u.host = "youtube.com".to_string(); - } -} diff --git a/server/src/url_rules/mod.rs b/server/src/url_rules/mod.rs index 03d53bd3e82d704a01ba3fd8dd02b7d31422c0de..9e1445346ce77a49dd6a7e7713bf9c57aef353cc 100644 --- a/server/src/url_rules/mod.rs +++ b/server/src/url_rules/mod.rs @@ -1,8 +1,12 @@ -//! URL canonicalization and hierarchy rules for [`crate::path_types::ItemId`]. +//! URL canonicalization and hierarchy via a semantic graph (DFA + generic fallback). -mod engine; +mod graph; +mod parse; mod registry; +#[cfg(test)] +mod registry_tests; + pub use registry::{ canonicalize_raw, looks_like_url, navigable_breadcrumbs, parent_url, resolve_id, CanonicalResult, }; diff --git a/server/src/url_rules/registry.rs b/server/src/url_rules/registry.rs index 14514e9af8385fb2b9b2f35eb9ee14d453d4b97c..8e6c012ea1fc74b864307bdacdf5a0f5db5259fc 100644 --- a/server/src/url_rules/registry.rs +++ b/server/src/url_rules/registry.rs @@ -1,12 +1,7 @@ -//! Per-domain canonicalization and hierarchy rules. +//! Public API: canonical identity and hierarchy via the URL graph. -use std::collections::HashSet; - -use super::engine::{ - clear_query, drop_fragment, drop_listing_suffix, force_https, keep_only_query, lowercase_host, - lowercase_path, normalize_reddit_host, normalize_youtube_host, rewrite_youtu_be, - rewrite_youtube_shorts, strip_tracking_params, strip_www, truncate_after_segment, ParsedUrl, -}; +use super::graph::graph; +use super::parse::UrlParts; /// Result of canonicalizing a raw URL string. #[derive(Debug, Clone, PartialEq, Eq)] @@ -16,71 +11,16 @@ pub struct CanonicalResult { pub alias_of: Option, } -fn apply_global(u: &mut ParsedUrl) { - force_https(u); - drop_fragment(u); - strip_www(u); - lowercase_host(u); - strip_tracking_params(u); -} - -fn normalize_reddit(u: &mut ParsedUrl) { - normalize_reddit_host(u); - lowercase_path(u); - truncate_after_segment(u, "comments", 1); - drop_listing_suffix(u, &["hot", "top", "new", "rising", "controversial"]); - clear_query(u); -} - -fn normalize_youtube(u: &mut ParsedUrl) { - rewrite_youtu_be(u); - normalize_youtube_host(u); - rewrite_youtube_shorts(u); - keep_only_query(u, &["v", "list"]); -} - -fn normalize_default(_u: &mut ParsedUrl) { - // Global rules only. -} - -fn domain_key(host: &str) -> &'static str { - if host == "reddit.com" || host.ends_with(".reddit.com") { - "reddit.com" - } else if host == "youtube.com" || host == "youtu.be" { - "youtube.com" - } else { - "default" - } -} - -fn normalize_for_host(u: &mut ParsedUrl) { - apply_global(u); - match domain_key(&u.host) { - "reddit.com" => normalize_reddit(u), - "youtube.com" => normalize_youtube(u), - _ => normalize_default(u), - } -} - -/// Structural path segments that must not become standalone tree nodes when more path follows. -fn structural_trailing(host: &str) -> &'static [&'static str] { - match domain_key(host) { - "reddit.com" => &["comments"], - _ => &[], - } -} - /// Canonicalize a raw URL. Returns `None` if the input is not URL-like. pub fn canonicalize_raw(raw: &str) -> Option { let trimmed = raw.trim(); if trimmed.is_empty() { return None; } - let mut u = ParsedUrl::parse(trimmed)?; - let input_snapshot = u.canonical_string()?; - normalize_for_host(&mut u); - let canonical = u.canonical_string()?; - let alias_of = if input_snapshot != canonical { + let parts = UrlParts::parse(trimmed)?; + let g = graph(); + let canonical = g.resolve_canonical(&parts)?; + let alias_of = if trimmed != canonical { Some(trimmed.to_string()) } else { None @@ -98,35 +38,14 @@ pub fn resolve_id(raw: &str) -> Option { /// Navigable ancestor URLs from domain root up to and including `canonical` (full URLs). pub fn navigable_breadcrumbs(canonical: &str) -> Vec { - let Some(u) = ParsedUrl::parse(canonical) else { - return vec![canonical.to_string()]; + let parts = match UrlParts::parse(canonical) { + Some(p) => p, + None => return vec![canonical.to_string()], }; - let structural: HashSet<&str> = structural_trailing(&u.host).iter().copied().collect(); - let n = u.path_segments.len(); - let mut out = Vec::new(); - - // Domain root (no path segments). - if let Some(base) = u.with_path_segments(&[]).canonical_string() { - out.push(base); - } - - for i in 0..n { - let segs: Vec = u.path_segments[..=i].to_vec(); - let is_last = i == n - 1; - let seg = u.path_segments[i].as_str(); - if structural.contains(seg) && !is_last { - continue; - } - if let Some(url) = u.with_path_segments(&segs).canonical_string() { - if out.last() != Some(&url) { - out.push(url); - } - } - } - out + graph().breadcrumbs(&parts) } -/// Immediate parent scope URL, or `None` for tree root / opaque single-segment ids. +/// Immediate parent scope URL, or `None` for tree root. pub fn parent_url(canonical: &str) -> Option { let crumbs = navigable_breadcrumbs(canonical); if crumbs.len() <= 1 { @@ -148,88 +67,3 @@ pub fn looks_like_url(raw: &str) -> bool { || t.starts_with("youtu.be/") } -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn reddit_post_drops_slug_and_normalizes_host() { - let r = canonicalize_raw( - "https://old.reddit.com/r/AmItheAsshole/comments/1trnvdl/aita_for_cancelling/", - ) - .unwrap(); - assert_eq!( - r.canonical, - "https://reddit.com/r/amitheasshole/comments/1trnvdl" - ); - } - - #[test] - fn reddit_strips_query_and_listing() { - assert_eq!( - canonicalize_raw("https://www.reddit.com/r/rust/?sort=top") - .unwrap() - .canonical, - "https://reddit.com/r/rust" - ); - assert_eq!( - canonicalize_raw("https://www.reddit.com/r/programming/hot") - .unwrap() - .canonical, - "https://reddit.com/r/programming" - ); - } - - #[test] - fn reddit_short_path() { - assert_eq!( - canonicalize_raw("r/rust").unwrap().canonical, - "https://reddit.com/r/rust" - ); - } - - #[test] - fn reddit_breadcrumbs_skip_phantom_comments() { - let post = "https://reddit.com/r/aww/comments/1trnvdl"; - let crumbs = navigable_breadcrumbs(post); - assert!(!crumbs.iter().any(|c| c.ends_with("/comments"))); - assert_eq!( - crumbs.last().map(String::as_str), - Some(post) - ); - assert!(crumbs.contains(&"https://reddit.com/r/aww".to_string())); - } - - #[test] - fn reddit_parent_of_post_is_subreddit() { - assert_eq!( - parent_url("https://reddit.com/r/aww/comments/1trnvdl").as_deref(), - Some("https://reddit.com/r/aww") - ); - } - - #[test] - fn youtube_youtu_be_and_watch_same_canonical() { - let a = canonicalize_raw("https://youtu.be/dQw4w9WgXcQ").unwrap().canonical; - let b = canonicalize_raw("https://www.youtube.com/watch?v=dQw4w9WgXcQ&t=10").unwrap(); - assert_eq!(a, b.canonical); - assert_eq!(a, "https://youtube.com/watch?v=dQw4w9WgXcQ"); - } - - #[test] - fn legacy_schemeless_upgrades() { - assert_eq!( - canonicalize_raw("reddit.com/r/rust/comments/aaa/announcing_rust_199") - .unwrap() - .canonical, - "https://reddit.com/r/rust/comments/aaa" - ); - } - - #[test] - fn alias_recorded_when_input_differs() { - let r = canonicalize_raw("https://youtu.be/abc123").unwrap(); - assert_eq!(r.canonical, "https://youtube.com/watch?v=abc123"); - assert!(r.alias_of.is_some()); - } -} Side B — contributor: tommy-mor Side B — commit message: [6e344666] Improve sorterc scan speed, errors, and ingest compile. Make scan a fast DSL parse pass with human-readable output, surface full parse_error details, and add compile --ingest for single-event replay from a log. Co-authored-by: Cursor Side B — unified diff (full patch): diff --git a/server/src/offline.rs b/server/src/offline.rs index 54ad0ded096a305ef8454ab2cdd1c3af71b14f5d..2db6f67e22e00a24ce673d13095644dd9fa9342d 100644 --- a/server/src/offline.rs +++ b/server/src/offline.rs @@ -28,6 +28,10 @@ pub struct CompileResult { pub threads: Vec, pub rankings: Vec, pub stats: CompileStats, + #[serde(skip_serializing_if = "Option::is_none")] + pub ingest_id: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub ingest_line: Option, } #[derive(Debug, Clone, Serialize)] @@ -36,6 +40,8 @@ pub struct CompileError { pub error: String, #[serde(skip_serializing_if = "Option::is_none")] pub hint: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub parse_error: Option, } #[derive(Debug, Clone, Serialize)] @@ -50,7 +56,7 @@ pub struct MalformedIngest { pub id: String, pub room_id: String, pub thread_tag: String, - pub reason: String, + pub parse_error: String, } #[derive(Debug, Clone, Serialize)] @@ -59,9 +65,36 @@ pub struct ScanResult { pub path: String, pub total_lines: usize, pub parsed_events: usize, + pub ingest_events: usize, pub bad_json_lines: Vec, pub malformed_ingests: Vec, - pub skipped_ingests: usize, +} + +#[derive(Debug)] +pub enum CompileIngestError { + NotFound(String), + Io(std::io::Error), + Compile(CompileError), +} + +impl CompileIngestError { + pub fn into_compile_error(self) -> CompileError { + match self { + Self::NotFound(id) => CompileError { + ok: false, + error: format!("ingest not found: {id}"), + hint: Some("pass the ingest event id from events.jsonl".into()), + parse_error: None, + }, + Self::Io(e) => CompileError { + ok: false, + error: format!("io error: {e}"), + hint: None, + parse_error: None, + }, + Self::Compile(e) => e, + } + } } fn document_stats(doc: &dsl::Document) -> CompileStats { @@ -166,19 +199,22 @@ fn rankings_for_simulated( .collect() } -/// Validate and simulate one `.sorter` document against optional base reducer state. -pub fn compile_document( +fn compile_document_inner( base: &ReducerState, room: &str, text: &str, + ingest_id: Option, + ingest_line: Option, ) -> Result { let room_key = room.trim(); let scope = scope_from_room_wire(room_key); let validated = validate_ingest_document(base, text, &scope).map_err(|(_, message, hint)| { + let parse_error = dsl::parse_full(text).err().map(|e| e.to_string()); CompileError { ok: false, error: message, hint, + parse_error, } })?; @@ -200,15 +236,27 @@ pub fn compile_document( threads: threads_in_document(text), rankings: rankings_for_simulated(&simulated, &scope, room_key, &validated.doc), stats: document_stats(&validated.doc), + ingest_id, + ingest_line, }) } +/// Validate and simulate one `.sorter` document against optional base reducer state. +pub fn compile_document( + base: &ReducerState, + room: &str, + text: &str, +) -> Result { + compile_document_inner(base, room, text, None, None) +} + fn ingest_parse_error(raw: &str) -> Option { dsl::parse_full(raw).err().map(|e| e.to_string()) } -fn load_events_from_jsonl(path: &Path) -> Result<(Vec<(usize, Event)>, Vec), std::io::Error> { +fn load_events_from_jsonl(path: &Path) -> Result<(usize, Vec<(usize, Event)>, Vec), std::io::Error> { let text = std::fs::read_to_string(path)?; + let total_lines = text.lines().count(); let mut events = Vec::new(); let mut bad_json_lines = Vec::new(); for (idx, line) in text.lines().enumerate() { @@ -225,61 +273,90 @@ fn load_events_from_jsonl(path: &Path) -> Result<(Vec<(usize, Event)>, Vec Result<(ReducerState, Vec), std::io::Error> { - let (events, bad_json_lines) = load_events_from_jsonl(path)?; +fn replay_events(events: &[(usize, Event)]) -> ReducerState { let mut state = ReducerState::default(); for (_line_no, ev) in events { - state.apply_event(ev); + state.apply_event(ev.clone()); } - Ok((state, bad_json_lines)) + state } -/// Scan an events.jsonl for corrupt JSON lines and ingests that fail DSL replay. -pub fn scan_jsonl(path: &Path) -> Result { - let text = std::fs::read_to_string(path)?; - let total_lines = text.lines().count(); - let (events, bad_json_lines) = load_events_from_jsonl(path)?; +/// Replay a JSONL event log into reducer state (same rules as server boot). +pub fn load_reducer_from_jsonl(path: &Path) -> Result<(ReducerState, Vec), std::io::Error> { + let (_total_lines, events, bad_json_lines) = load_events_from_jsonl(path)?; + Ok((replay_events(&events), bad_json_lines)) +} - let mut malformed_ingests = Vec::new(); - let mut skipped_ingests = 0usize; - let mut state = ReducerState::default(); - let parsed_events = events.len(); +/// Find one ingest in a log and compile it against all prior events as base state. +pub fn compile_ingest_from_log(path: &Path, ingest_id: &str) -> Result { + let (_total_lines, events, bad_json_lines) = load_events_from_jsonl(path).map_err(CompileIngestError::Io)?; + if !bad_json_lines.is_empty() { + return Err(CompileIngestError::Compile(CompileError { + ok: false, + error: format!("jsonl has {} corrupt line(s)", bad_json_lines.len()), + hint: Some("fix the log or use `sorterc scan`".into()), + parse_error: None, + })); + } + + let needle = ingest_id.trim(); + let mut found: Option<(usize, Ingest)> = None; + let mut prior: Vec<(usize, Event)> = Vec::new(); for (line_no, ev) in events { if let Event::Ingest(ref ing) = ev { - if let Some(reason) = ingest_parse_error(&ing.raw) { + if ing.id == needle { + found = Some((line_no, ing.clone())); + break; + } + } + prior.push((line_no, ev)); + } + + let (line_no, ing) = found.ok_or_else(|| CompileIngestError::NotFound(needle.to_string()))?; + let base = replay_events(&prior); + compile_document_inner(&base, &ing.room_id, &ing.raw, Some(ing.id.clone()), Some(line_no)) + .map_err(CompileIngestError::Compile) +} + +/// Scan an events.jsonl for corrupt JSON lines and ingests whose DSL fails to parse. +/// +/// This does not replay the log (which would run rank centrality on every ingest and +/// can take minutes on real logs). It matches what the server skips on boot: parse failure. +pub fn scan_jsonl(path: &Path) -> Result { + let (total_lines, events, bad_json_lines) = load_events_from_jsonl(path)?; + + let mut malformed_ingests = Vec::new(); + let mut ingest_events = 0usize; + + for (line_no, ev) in &events { + if let Event::Ingest(ing) = ev { + ingest_events += 1; + if let Some(parse_error) = ingest_parse_error(&ing.raw) { malformed_ingests.push(MalformedIngest { - line: line_no, + line: *line_no, id: ing.id.clone(), room_id: ing.room_id.clone(), thread_tag: ing.thread_tag.clone(), - reason, + parse_error, }); } - let before = state.ingests_by_id.len(); - state.apply_event(ev); - if state.ingests_by_id.len() == before { - skipped_ingests += 1; - } - } else { - state.apply_event(ev); } } - let ok = bad_json_lines.is_empty() && malformed_ingests.is_empty() && skipped_ingests == 0; + let ok = bad_json_lines.is_empty() && malformed_ingests.is_empty(); Ok(ScanResult { ok, path: path.display().to_string(), total_lines, - parsed_events, + parsed_events: events.len(), + ingest_events, bad_json_lines, malformed_ingests, - skipped_ingests, }) } @@ -330,4 +407,25 @@ mod tests { assert!(!report.ok); assert_eq!(report.bad_json_lines.len(), 1); } + + #[test] + fn scan_reports_dsl_parse_error_detail() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("events.jsonl"); + let ingest = serde_json::json!({ + "type": "ingest", + "ts": 1, + "id": "bad-ingest-id", + "raw": "{ no closing brace\n~/a { body }\n~/a 1:0 ~/b", + "principal": "test", + "room_id": "public", + "thread_tag": "t", + }); + std::fs::write(&path, format!("{ingest}\n")).unwrap(); + let report = scan_jsonl(&path).unwrap(); + assert!(!report.ok); + assert_eq!(report.malformed_ingests.len(), 1); + assert_eq!(report.malformed_ingests[0].id, "bad-ingest-id"); + assert!(report.malformed_ingests[0].parse_error.contains("parse error")); + } } diff --git a/sorterc/readme.md b/sorterc/readme.md index 1ebcc3fc935541ea9e47e0458ec67a750fe19fff..e388615c34b40914ca51fc7a79cb738eebc2909b 100644 --- a/sorterc/readme.md +++ b/sorterc/readme.md @@ -60,21 +60,33 @@ Rankings use the same structure as the server's dry-run check: parent scope, con ### `scan` — lint an `events.jsonl` -Reads a JSONL event log and reports problems without starting a server. +Fast single-pass check. Does **not** replay the log (replay runs rank centrality on every ingest and gets slow fast). ```bash cargo run -p sorterc -- scan events.jsonl -cargo run -p sorterc -- scan events.jsonl --pretty +cargo run -p sorterc -- scan events.jsonl --json +cargo run -p sorterc -- scan events.jsonl --json --pretty ``` +Default output is human-readable with a blank line between each problem. Use `--json` for machine output. + Reports: - **bad JSON lines** — lines that are not valid JSON -- **malformed ingests** — ingest events whose `raw` DSL fails to parse -- **skipped ingests** — ingests dropped during replay (same behavior as server boot) +- **malformed ingests** — ingest events whose `raw` DSL fails to parse, with full `parse_error` text Exits 0 when clean, 1 when any issue is found. +### `compile --ingest` — compile one event from a log + +Replay all events **before** the target ingest as base state, then compile that ingest's DSL: + +```bash +cargo run -p sorterc -- compile --ingest cabd8adc-57ae-402d-a940-8e24339ac451 --from events.jsonl --pretty +``` + +Output includes `ingest_id`, `ingest_line`, and rankings for that post only. This may take a while for ingests late in a large log (full replay up to that point). + ## Typical uses - Iterate on `.sorter` files in an editor and pipe through `compile` to see rankings instantly diff --git a/sorterc/src/main.rs b/sorterc/src/main.rs index 71382c180085cb0ad71043c852f8db5d3a48a284..5687c850d16973d28749380b8dc89f3bcebbfc56 100644 --- a/sorterc/src/main.rs +++ b/sorterc/src/main.rs @@ -22,9 +22,15 @@ struct Cli { enum Command { /// Parse and simulate a .sorter document; emit ranking JSON to stdout. Compile { - /// `.sorter` file, or `-` for stdin. - file: PathBuf, - /// Room wire id (`public` or private room id). + /// `.sorter` file, or `-` for stdin. Omit when using --ingest. + file: Option, + /// Compile one ingest event from a log (by event id / uuid). + #[arg(long)] + ingest: Option, + /// events.jsonl containing the ingest (required with --ingest). + #[arg(long)] + from: Option, + /// Room wire id (`public` or private room id). Ignored with --ingest. #[arg(long, default_value = "public")] room: String, /// Optional events.jsonl to replay before compiling (seed garden state). @@ -34,9 +40,13 @@ enum Command { #[arg(long)] pretty: bool, }, - /// Scan an events.jsonl for corrupt JSON lines and malformed ingests. + /// Scan an events.jsonl for corrupt JSON lines and DSL parse failures. Scan { file: PathBuf, + /// Emit JSON instead of human-readable output. + #[arg(long)] + json: bool, + /// Pretty-print JSON (requires --json). #[arg(long)] pretty: bool, }, @@ -58,6 +68,7 @@ fn load_base_state(base: Option<&Path>) -> Result { let Some(path) = base else { return Ok(ReducerState::default()); }; + eprintln!("sorterc: replaying {} for base state (may take a while on large logs)…", path.display()); let (state, bad_lines) = offline::load_reducer_from_jsonl(path) .with_context(|| format!("load base jsonl {}", path.display()))?; if !bad_lines.is_empty() { @@ -78,25 +89,101 @@ fn print_json(value: &T, pretty: bool) -> Result<()> { Ok(()) } -fn run_compile(file: PathBuf, room: String, base: Option, pretty: bool) -> Result<()> { - let text = read_input(&file)?; - let base_state = load_base_state(base.as_deref())?; - match offline::compile_document(&base_state, &room, &text) { - Ok(result) => { - print_json::(&result, pretty)?; - Ok(()) +fn run_compile( + file: Option, + ingest: Option, + from: Option, + room: String, + base: Option, + pretty: bool, +) -> Result<()> { + if ingest.is_some() ^ from.is_some() { + bail!("--ingest and --from must be used together"); + } + if ingest.is_some() && (file.is_some() || base.is_some()) { + bail!("with --ingest/--from, omit file and --base"); + } + if file.is_none() && ingest.is_none() { + bail!("pass a .sorter file or --ingest --from events.jsonl"); + } + + if let (Some(ingest_id), Some(log_path)) = (ingest, from) { + eprintln!( + "sorterc: compiling ingest {ingest_id} from {}…", + log_path.display() + ); + match offline::compile_ingest_from_log(&log_path, &ingest_id) { + Ok(result) => { + print_json::(&result, pretty)?; + Ok(()) + } + Err(err) => { + print_json::(&err.into_compile_error(), pretty)?; + std::process::exit(1); + } + } + } else { + let file = file.expect("checked above"); + let text = read_input(&file)?; + let base_state = load_base_state(base.as_deref())?; + match offline::compile_document(&base_state, &room, &text) { + Ok(result) => { + print_json::(&result, pretty)?; + Ok(()) + } + Err(err) => { + print_json::(&err, pretty)?; + std::process::exit(1); + } + } + } +} + +fn print_scan_human(report: &ScanResult) { + if report.ok { + println!( + "ok: {} ({} lines, {} events, {} ingests)", + report.path, report.total_lines, report.parsed_events, report.ingest_events + ); + return; + } + + let problems = report.bad_json_lines.len() + report.malformed_ingests.len(); + println!( + "{}: {} lines, {} events, {} ingests, {problems} problem(s)", + report.path, report.total_lines, report.parsed_events, report.ingest_events + ); + + let mut first = true; + for bad in &report.bad_json_lines { + if !first { + println!(); } - Err(err) => { - print_json::(&err, pretty)?; - std::process::exit(1); + first = false; + println!("line {}: invalid JSON", bad.line); + println!("{}", bad.message); + } + + for bad in &report.malformed_ingests { + if !first { + println!(); } + first = false; + println!( + "line {}: ingest {} (#{} in {})", + bad.line, bad.id, bad.thread_tag, bad.room_id + ); + println!("{}", bad.parse_error); } } -fn run_scan(file: PathBuf, pretty: bool) -> Result<()> { - let report = offline::scan_jsonl(&file) - .with_context(|| format!("scan {}", file.display()))?; - print_json::(&report, pretty)?; +fn run_scan(file: PathBuf, json: bool, pretty: bool) -> Result<()> { + let report = offline::scan_jsonl(&file).with_context(|| format!("scan {}", file.display()))?; + if json { + print_json::(&report, pretty)?; + } else { + print_scan_human(&report); + } if !report.ok { std::process::exit(1); } @@ -108,10 +195,12 @@ fn main() -> Result<()> { match cli.cmd { Command::Compile { file, + ingest, + from, room, base, pretty, - } => run_compile(file, room, base, pretty), - Command::Scan { file, pretty } => run_scan(file, pretty), + } => run_compile(file, ingest, from, room, base, pretty), + Command::Scan { file, json, pretty } => run_scan(file, json, pretty), } }