You are a constitutional council ranking individual git commits for ownership allocation. Compare these two commits. Decide which contributed more lasting value to the project. Judge substance, not spectacle: - Prefer correct, lasting design and real bugfixes over churn, formatting, renames, or generated noise. - Prefer clarity and necessity over sheer line count. A small precise change can beat a large diffuse one. - Do not favor a side merely because its patch is longer or noisier. - Weight what the change does for the project, not the contributor's name. Return ONLY a JSON object: {"winner": "A" or "B", "ratio": "N:M", "explanation": "..."} The explanation must cite concrete differences in the patches (1-3 sentences). Side A — contributor: tommy-mor Side A — commit message: [c7ef287e] Rank every eligible commit with the LLM council. Stop short-circuiting on a single contributor; pairwise-sort commits, roll scores up for emission payouts, and surface commit rankings on epoch pages. Co-authored-by: Cursor Side A — unified diff (full patch): diff --git a/constitution.py b/constitution.py index 26ba130e885e0e69fb7874ca5c3f07f42100a150..71fc46b4c9a4d7a9ea7bf319860b0db1acc4a673 100644 --- a/constitution.py +++ b/constitution.py @@ -503,20 +503,22 @@ def _epochs_in_ledger() -> list[int]: def build_pairwise_prompt(side_a: dict, side_b: dict) -> str: - return f"""You are ranking contributions to an open source project. -Compare these two sides (each may be one or more commits). Decide which side contributed more. + return f"""You are ranking individual git commits to an open source project. +Compare these two commits. Decide which commit contributed more. Return ONLY a JSON object: {{"winner": "A" or "B", "ratio": "N:M", "explanation": "..."}} -Side A — commit messages: +Side A — contributor: {side_a.get('contributor', '?')} +Side A — commit message: {side_a['message']} -Side A — unified diffs (full patches): +Side A — unified diff (full patch): {side_a['diff']} -Side B — commit messages: +Side B — contributor: {side_b.get('contributor', '?')} +Side B — commit message: {side_b['message']} -Side B — unified diffs (full patches): +Side B — unified diff (full patch): {side_b['diff']}""" @@ -1518,16 +1520,28 @@ async def broadcast_js(js: str): await queue.put(js) -def _author_side_for_llm(author: str, author_commits: dict) -> dict: - cs = author_commits[author] +def _commit_side_for_llm(row: dict) -> dict: + oid = row["oid"] + short = oid.split(":", 1)[1][:8] if ":" in oid else oid[:8] return { - "message": "\n".join(f"[{c['sha']}] {c['message']}" for c in cs), - "diff": "\n\n".join(f"=== {c['sha']} ===\n{c['diff']}" for c in cs), - "commit_ids": [c["commit_id"] for c in cs], - "contributor": author, + "message": f"[{short}] {row['message']}", + "diff": row["patch"] or "", + "commit_id": commit_id_for_oid(oid), + "contributor": row["contributor"], + "oid": oid, } +def _rollup_contributor_scores( + ordered: list[dict], commit_scores: list[Decimal] +) -> dict[str, Decimal]: + totals: dict[str, Decimal] = {} + for row, score in zip(ordered, commit_scores): + contributor = row["contributor"] + totals[contributor] = totals.get(contributor, Decimal("0")) + score + return totals + + def _find_judgment(comparison_id: str, model_id: str) -> dict | None: for e in evidence_by_kind("llm.judgment"): p = e.payload @@ -1546,101 +1560,106 @@ def _find_ranking_models(ranking_run_id: str) -> list[str] | None: async def rank_commits(commits: list[dict], *, epoch: int = -1): + """Pairwise-rank every eligible commit; roll scores up to contributors.""" if not commits: return {}, [], {"ranking_run_id": "", "ranking_event_id": ""} - commit_ids = sorted(commit_id_for_oid(row["oid"]) for row in commits) + ordered = sorted(commits, key=lambda r: r["oid"]) + commit_ids = [commit_id_for_oid(row["oid"]) for row in ordered] ranking_run_id = _content_id("rank", { "epoch": epoch, - "commit_ids": commit_ids, + "commit_ids": sorted(commit_ids), }) - contributors = sorted(set(c["contributor"] for c in commits)) + contributors = sorted({c["contributor"] for c in ordered}) - if len(contributors) == 1: + # Nothing to compare: a single commit (not a single contributor). + if len(ordered) == 1: await append_evidence(epoch, "ranking.started", { "ranking_run_id": ranking_run_id, "commit_ids": commit_ids, "contributors": contributors, "models": [], - "summary": f"ranking epoch {epoch}: single contributor", + "summary": f"ranking epoch {epoch}: single commit", }) - ranking = {contributors[0]: Decimal("1")} + commit_ranking = {commit_ids[0]: "1"} + contributor_ranking = {ordered[0]["contributor"]: Decimal("1")} completed = await append_evidence(epoch, "ranking.completed", { "ranking_run_id": ranking_run_id, "models": [], - "ranking": {contributors[0]: "1"}, + "commit_ranking": commit_ranking, + "contributor_ranking": {ordered[0]["contributor"]: "1"}, + "ranking": {ordered[0]["contributor"]: "1"}, "judgment_ids": [], - "summary": f"Only {contributors[0]} is eligible; rank is 1.0", + "summary": f"Only one eligible commit; {ordered[0]['contributor']} rank 1.0", }) await broadcast_audit( "ranking", - f"Only {contributors[0]} is eligible; rank is 1.0", + f"Only one eligible commit; {ordered[0]['contributor']} rank 1.0", progress=90, phase="finalizing", evidence_event_id=completed.event_id, evidence_url=_evidence_url("event", completed.event_id), links={"epoch": _evidence_url("epoch", str(epoch))}, ) - return ranking, [], { + return contributor_ranking, [], { "ranking_run_id": ranking_run_id, "ranking_event_id": completed.event_id, } if not (OPENROUTER_API_KEY or "").strip(): raise RuntimeError( - "OPENROUTER_API_KEY is required when multiple contributors need ranking" + "OPENROUTER_API_KEY is required when multiple commits need ranking" ) models = _find_ranking_models(ranking_run_id) if models is None: models = await fetch_top_models(n=3) if not models: - raise RuntimeError("no council models available for contributor ranking") + raise RuntimeError("no council models available for commit ranking") await append_evidence(epoch, "ranking.started", { "ranking_run_id": ranking_run_id, "commit_ids": commit_ids, "contributors": contributors, "models": models, - "summary": f"Council selected: {', '.join(models)}", + "summary": ( + f"Council selected: {', '.join(models)} — " + f"{len(ordered)} commits" + ), }) await broadcast_audit( "council", - f"Council selected: {', '.join(models)}", + f"Council selected: {', '.join(models)} — ranking {len(ordered)} commits", progress=35, phase="ranking", ) await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ - ["div.log-council", f"Council: {', '.join(models)} — {len(commits)} commits"] + ["div.log-council", + f"Council: {', '.join(models)} — {len(ordered)} commits"] ])) - authors = contributors - author_commits = {a: [] for a in authors} - for row in sorted(commits, key=lambda r: r["oid"]): - author_commits[row["contributor"]].append({ - "message": row["message"], - "sha": row["oid"].split(":", 1)[1][:8], - "diff": row["patch"], - "commit_id": commit_id_for_oid(row["oid"]), - }) - + sides = [_commit_side_for_llm(row) for row in ordered] judgment_ids: list[str] = [] async def compare_fn(i, j): - a1, a2 = authors[i], authors[j] - side_a = _author_side_for_llm(a1, author_commits) - side_b = _author_side_for_llm(a2, author_commits) + side_a, side_b = sides[i], sides[j] + label_a = f"{side_a['commit_id'][:16]} ({side_a['contributor']})" + label_b = f"{side_b['commit_id'][:16]} ({side_b['contributor']})" prompt = build_pairwise_prompt(side_a, side_b) comparison_material = { "ranking_run_id": ranking_run_id, "side_a": { - "contributor": a1, - "commit_ids": side_a["commit_ids"], + "contributor": side_a["contributor"], + "commit_id": side_a["commit_id"], + "commit_ids": [side_a["commit_id"]], + "oid": side_a["oid"], "message": _bytes_blob(side_a["message"]), "diff": _bytes_blob(side_a["diff"]), }, "side_b": { - "contributor": a2, - "commit_ids": side_b["commit_ids"], + "contributor": side_b["contributor"], + "commit_id": side_b["commit_id"], + "commit_ids": [side_b["commit_id"]], + "oid": side_b["oid"], "message": _bytes_blob(side_b["message"]), "diff": _bytes_blob(side_b["diff"]), }, @@ -1650,22 +1669,24 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1): comparison_material = { **comparison_material, "comparison_id": comparison_id, - "summary": f"Comparing {a1} with {a2}", + "summary": f"Comparing {label_a} with {label_b}", } cmp_ev = await append_evidence(epoch, "comparison.input", comparison_material) await broadcast_audit( "comparison", - f"Comparing {a1} with {a2}", + f"Comparing commits {label_a} vs {label_b}", phase="ranking", evidence_event_id=cmp_ev.event_id, evidence_url=_evidence_url("comparison", comparison_id), links={ "comparison": _evidence_url("comparison", comparison_id), + "commit_a": _evidence_url("commit", side_a["commit_id"]), + "commit_b": _evidence_url("commit", side_b["commit_id"]), "epoch": _evidence_url("epoch", str(epoch)), }, ) await broadcast_js(exec_event(Three[Selector("#emission-status")][MORPH][ - ["div#emission-status", f"Comparing {a1} vs {a2}…"] + ["div#emission-status", f"Comparing {label_a} vs {label_b}…"] ])) results = [] for model in models: @@ -1697,9 +1718,15 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1): jud_id = (existing or _find_judgment(comparison_id, model) or {}).get( "judgment_id" ) + win_label = ( + f"{sides[w]['commit_id'][:16]} ({sides[w]['contributor']})" + ) + lose_label = ( + f"{sides[l]['commit_id'][:16]} ({sides[l]['contributor']})" + ) await broadcast_audit( "vote", - f"{model}: {authors[w]} over {authors[l]} ({result['ratio']})", + f"{model}: {win_label} over {lose_label} ({result['ratio']})", phase="ranking", evidence_url=( _evidence_url("judgment", jud_id) if jud_id else None @@ -1714,8 +1741,8 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1): await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ ["div.log-vote", ["span.model", model], " — ", - ["span.winner", authors[w]], f" beat ", - ["span.loser", authors[l]], f" ({result['ratio']}) ", + ["span.winner", win_label], f" beat ", + ["span.loser", lose_label], f" ({result['ratio']}) ", ["span.explanation", result["explanation"]], ] ])) @@ -1745,24 +1772,46 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1): ["div#emission-status", label] ])) - pairs = await pairwise_rank(len(authors), compare_fn, progress_fn) + pairs = await pairwise_rank(len(ordered), compare_fn, progress_fn) if not pairs: - ranking = {authors[0]: Decimal("1")} if authors else {} + commit_score_list = [Decimal("1")] else: scores = rank_centrality(pairs) - ranking = {authors[i]: Decimal(str(scores[i])) for i in range(len(authors))} - ranking_rows = sorted(ranking.items(), key=lambda x: x[1], reverse=True) + commit_score_list = [Decimal(str(scores[i])) for i in range(len(ordered))] + + commit_ranking = { + commit_ids[i]: str(commit_score_list[i]) for i in range(len(ordered)) + } + contributor_totals = _rollup_contributor_scores(ordered, commit_score_list) + contrib_rows = sorted( + contributor_totals.items(), key=lambda x: x[1], reverse=True + ) + commit_rows = sorted( + ((commit_ids[i], commit_score_list[i], ordered[i]["contributor"]) + for i in range(len(ordered))), + key=lambda x: x[1], + reverse=True, + ) completed = await append_evidence(epoch, "ranking.completed", { "ranking_run_id": ranking_run_id, "models": models, - "ranking": {a: str(s) for a, s in ranking_rows}, + "commit_ranking": commit_ranking, + "contributor_ranking": {a: str(s) for a, s in contrib_rows}, + "ranking": {a: str(s) for a, s in contrib_rows}, "judgment_ids": judgment_ids, - "summary": "Ranking: " + ", ".join(f"{a} {s:.4f}" for a, s in ranking_rows), + "summary": ( + "Commit ranking: " + + ", ".join( + f"{cid[:16]}={float(s):.4f}" for cid, s, _ in commit_rows[:12] + ) + + ("…" if len(commit_rows) > 12 else "") + ), }) await broadcast_audit( "ranking", - "Ranking: " + ", ".join(f"{a} {s:.4f}" for a, s in ranking_rows), + "Contributor rollup: " + + ", ".join(f"{a} {s:.4f}" for a, s in contrib_rows), progress=90, phase="finalizing", evidence_event_id=completed.event_id, @@ -1772,10 +1821,10 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1): await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][ ["div.log-ranking", ["b", "Ranking: "], - *[["span.rank-entry", f"{a} {float(s):.3f} "] for a, s in ranking_rows], + *[["span.rank-entry", f"{a} {float(s):.3f} "] for a, s in contrib_rows], ] ])) - return ranking, models, { + return contributor_totals, models, { "ranking_run_id": ranking_run_id, "ranking_event_id": completed.event_id, } @@ -2352,6 +2401,23 @@ async def epoch_detail(epoch: int): ranking_nodes: list = [] if ranking_completed: + commit_ranking = ranking_completed.payload.get("commit_ranking") or {} + contrib_ranking = ( + ranking_completed.payload.get("contributor_ranking") + or ranking_completed.payload.get("ranking") + or {} + ) + commit_rank_links = [ + ( + f"{cid[:20]} = {score}", + _evidence_path("commit", cid), + ) + for cid, score in sorted( + commit_ranking.items(), + key=lambda kv: Decimal(str(kv[1])), + reverse=True, + ) + ] ranking_nodes = [ ["p", ranking_completed.payload.get("summary") or "ranking completed"], _dl_rows([ @@ -2362,10 +2428,10 @@ async def epoch_detail(epoch: int): ranking_completed.event_id, )), ]), - ["pre.blob", json.dumps( - ranking_completed.payload.get("ranking") or {}, - indent=2, sort_keys=True, - )], + ["h3", "commit ranking"], + _link_list(commit_rank_links) if commit_rank_links else ["p.note", "(none)"], + ["h3", "contributor rollup"], + ["pre.blob", json.dumps(contrib_ranking, indent=2, sort_keys=True)], ] elif ranking_started: ranking_nodes = [["p.note", f"Ranking started: {ranking_started.event_id}"]] @@ -2386,14 +2452,9 @@ async def epoch_detail(epoch: int): else: emission_node = ["p.note", "No emission for this epoch."] - contributors = { - e.payload.get("contributor") - for e in commit_evs - if e.payload.get("contributor") - } no_comparisons_note = "No comparisons." - if len(contributors) <= 1: - no_comparisons_note += " Single-contributor — no LLM judgments." + if len(commit_evs) <= 1: + no_comparisons_note += " Fewer than two eligible commits — nothing to pairwise-rank." body = [ _evidence_nav(), @@ -2511,9 +2572,14 @@ async def comparison_detail(comparison_id: str): j.payload.get("summary") or jid, _evidence_path("judgment", jid), )) - commit_links = [] - for cid in (side_a.get("commit_ids") or []) + (side_b.get("commit_ids") or []): - commit_links.append((cid, _evidence_path("commit", cid))) + commit_ids = [] + for side in (side_a, side_b): + cid = side.get("commit_id") + if cid: + commit_ids.append(cid) + else: + commit_ids.extend(side.get("commit_ids") or []) + commit_links = [(cid, _evidence_path("commit", cid)) for cid in commit_ids] return _evidence_page(f"comparison {comparison_id[:24]}", [ _evidence_nav(_a(_evidence_path("epoch", str(ev.epoch)), f"epoch {ev.epoch}")), ["div.eyebrow", "comparison"], @@ -2522,8 +2588,8 @@ async def comparison_detail(comparison_id: str): ("summary", p.get("summary")), ("ranking_run_id", p.get("ranking_run_id")), ("evidence_event", _a(_evidence_path("event", ev.event_id), ev.event_id)), - ("side_a", side_a.get("contributor")), - ("side_b", side_b.get("contributor")), + ("side_a", f"{side_a.get('commit_id', '?')} ({side_a.get('contributor', '?')})"), + ("side_b", f"{side_b.get('commit_id', '?')} ({side_b.get('contributor', '?')})"), ]), ["p", _a(f"/comparisons/{comparison_id}/prompt", "download prompt")], ["h2", "commits"], @@ -2534,12 +2600,12 @@ async def comparison_detail(comparison_id: str): _link_list(judgment_links), ["h2", "prompt"], _pre_blob(_blob_text(p.get("prompt"))), - ["h2", f"side A — {side_a.get('contributor', '?')}"], + ["h2", f"side A — {side_a.get('commit_id', side_a.get('contributor', '?'))}"], ["h3", "message"], _pre_blob(_blob_text(side_a.get("message"))), ["h3", "diff"], _pre_blob(_blob_text(side_a.get("diff"))), - ["h2", f"side B — {side_b.get('contributor', '?')}"], + ["h2", f"side B — {side_b.get('commit_id', side_b.get('contributor', '?'))}"], ["h3", "message"], _pre_blob(_blob_text(side_b.get("message"))), ["h3", "diff"], @@ -3156,9 +3222,9 @@ async def watch(): ], ["p.note", ( - "Pairwise council voting is ready." + "Pairwise council ranks every eligible commit." if key_ok else - "Single-contributor epochs can finalize, but contested rankings require OPENROUTER_API_KEY." + "Epochs with two or more eligible commits require OPENROUTER_API_KEY." ) ], ], diff --git a/tests/integration.clj b/tests/integration.clj index 11140e95a8cd86768a661a30e6b29a3ddfc95fcc..39b2476cb5ddf80733769f61343c81ba7287a974 100644 --- a/tests/integration.clj +++ b/tests/integration.clj @@ -507,7 +507,7 @@ (bind or-state @(:state or-mock)) (assert! (pos? (:model-requests or-state)) "OpenRouter /models was called") - (assert! (>= (:compare-requests or-state) 3) "at least 3 pairwise LLM calls (2 authors × 3 models)") + (assert! (>= (:compare-requests or-state) 3) "at least 3 pairwise LLM calls (2 commits × 3 models)") (bind ledger2 (get-json base-url "/api/ledger")) (assert! (>= (count ledger2) 3) @@ -543,6 +543,8 @@ "epoch page links comparisons") (assert! (str/includes? epoch-html "/judgments/") "epoch page links judgments") + (assert! (str/includes? epoch-html "commit ranking") + "epoch page shows per-commit ranking") (bind commit-href (second (re-find #"/commits/(c_[a-f0-9]+)" epoch-html))) (assert! (some? commit-href) "found a commit id on epoch page") diff --git a/tests/test_git_discovery.py b/tests/test_git_discovery.py index 00d26bfa85737b178c8822804e278211974e0efe..0f002a58bd9a122d41c5a85e32c16ed2b4404d2f 100644 --- a/tests/test_git_discovery.py +++ b/tests/test_git_discovery.py @@ -434,7 +434,7 @@ def test_emission_distribution_sums_exactly_to_total( assert entry.discovery_snapshot_id == "ranked-snapshot" -def test_single_contributor_ranking_is_total_and_uses_no_pairwise_votes( +def test_single_commit_ranking_skips_pairwise( discovery_config, monkeypatch, ): monkeypatch.setattr(c, "store", c.JsonlStore(discovery_config / "ledger.jsonl")) @@ -454,6 +454,65 @@ def test_single_contributor_ranking_is_total_and_uses_no_pairwise_votes( assert info["ranking_event_id"] +def test_same_contributor_multiple_commits_runs_pairwise( + discovery_config, monkeypatch, +): + monkeypatch.setattr(c, "store", c.JsonlStore(discovery_config / "ledger.jsonl")) + calls = {"n": 0} + + async def models(n=3): + return ["m1", "m2", "m3"] + + async def compare(model_id, side_a, side_b, **kwargs): + calls["n"] += 1 + assert "commit_id" in side_a and "commit_id" in side_b + if kwargs.get("persist"): + attempt_id = c.attempt_id_for(kwargs["comparison_id"], model_id, 1) + await c.append_evidence(kwargs["epoch"], "llm.judgment", { + "judgment_id": c.judgment_id_for({ + "attempt_id": attempt_id, + "comparison_id": kwargs["comparison_id"], + "model_id": model_id, + "winner": "A", + "ratio": "2:1", + "explanation": "ok", + }), + "attempt_id": attempt_id, + "comparison_id": kwargs["comparison_id"], + "model_id": model_id, + "winner": "A", + "ratio": "2:1", + "explanation": "ok", + "summary": "ok", + }) + return {"winner": "A", "ratio": "2:1", "explanation": "ok"} + + monkeypatch.setattr(c, "fetch_top_models", models) + monkeypatch.setattr(c, "llm_pairwise_compare", compare) + monkeypatch.setattr(c, "OPENROUTER_API_KEY", "test-key") + commits = [ + { + "contributor": "alice", + "oid": "sha1:" + char * 40, + "message": f"msg-{char}", + "patch": f"patch-{char}", + } + for char in ("a", "b", "c") + ] + ranking, used, info = asyncio.run(c.rank_commits(commits, epoch=0)) + assert set(ranking) == {"alice"} + assert ranking["alice"] > 0 + assert used == ["m1", "m2", "m3"] + assert calls["n"] >= 3 + completed = next( + e for e in c.store.read() + if isinstance(e, c.Evidence) and e.kind == "ranking.completed" + ) + assert len(completed.payload["commit_ranking"]) == 3 + assert "alice" in completed.payload["contributor_ranking"] + assert info["ranking_event_id"] + + def test_any_council_failure_aborts_ranking(discovery_config, monkeypatch): monkeypatch.setattr(c, "store", c.JsonlStore(discovery_config / "ledger.jsonl")) @@ -479,16 +538,16 @@ def test_any_council_failure_aborts_ranking(discovery_config, monkeypatch): asyncio.run(c.rank_commits(commits, epoch=0)) -def test_contested_ranking_requires_openrouter_key(monkeypatch): +def test_multi_commit_ranking_requires_openrouter_key(monkeypatch): monkeypatch.setattr(c, "OPENROUTER_API_KEY", "") commits = [ { - "contributor": contributor, + "contributor": "alice", "oid": "sha1:" + char * 40, - "message": contributor, + "message": char, "patch": "patch", } - for contributor, char in [("alice", "a"), ("bob", "b")] + for char in ("a", "b") ] with pytest.raises(RuntimeError, match="OPENROUTER_API_KEY"): asyncio.run(c.rank_commits(commits)) Side B — contributor: tommy-mor Side B — commit message: [8d8230d1] reddit Side B — unified diff (full patch): diff --git a/.gitignore b/.gitignore index 4c7073f9fac0c30fd2050d79a60ef447af58ebeb..ada462e900d24a3a6d08165d158c80f79c35a5a5 100644 --- a/.gitignore +++ b/.gitignore @@ -7,3 +7,4 @@ data/ repomix-output.xml dev-data/ +.env diff --git a/server/Cargo.toml b/server/Cargo.toml index 7906a8547d56b8e6a48ef59c37aa82a8510fdee9..4677fedcb45292eebebe7e9cf6ce2f5738f18ddf 100644 --- a/server/Cargo.toml +++ b/server/Cargo.toml @@ -16,6 +16,7 @@ tower = "0.5" tower-http = { version = "0.5", features = ["trace"] } tracing = "0.1" tracing-subscriber = { version = "0.3", features = ["env-filter"] } +reqwest = { version = "0.12", features = ["json"] } [dev-dependencies] reqwest = { version = "0.12", features = ["json"] } diff --git a/server/src/html/mod.rs b/server/src/html/mod.rs index 2864407ed6e8ec284a1dc663acf1805534566b62..df6505021d9f446c2b453e20e3eb3cf696a111f9 100644 --- a/server/src/html/mod.rs +++ b/server/src/html/mod.rs @@ -272,5 +272,16 @@ pub async fn home(State(state): State, uri: Uri) -> impl IntoResponse pub async fn browse(State(state): State, uri: Uri) -> impl IntoResponse { let item = ItemId::from_browse_uri(uri.path()).unwrap_or(ItemId::root()); + if item.as_str().starts_with("reddit.com") { + let needs_fetch = { + let tree = state.tree.read().await; + tree.get(&item) + .map(|n| n.data.is_none()) + .unwrap_or(true) + }; + if needs_fetch { + state.reddit.request_fetch(item.clone()); + } + } item_page(state, uri, item).await } diff --git a/server/src/reddit.rs b/server/src/reddit.rs index d203dca09245daf869b3aa942898447700ae69fb..90053ad03b1d7c8e94f325dd4ee64c2b4f7da900 100644 --- a/server/src/reddit.rs +++ b/server/src/reddit.rs @@ -1,4 +1,12 @@ -//! Reddit API import (async, decoupled from UI request path). +//! Reddit API import via a single background worker (rate limits, dedup, backoff). + +use std::collections::{HashMap, HashSet}; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use reqwest::{header, Client, StatusCode}; +use serde::Deserialize; +use tokio::sync::{mpsc, RwLock}; use crate::{ path_types::ItemId, @@ -10,12 +18,401 @@ pub fn ensure_partial_tree(tree: &mut GlobalTree, id: &ItemId) { tree.ensure_path(id); } -/// Placeholder for Reddit JSON import. Returns entity data when implemented. -pub async fn fetch_reddit_entity(_id: &ItemId) -> Option { - None +pub struct RedditCommand { + pub id: ItemId, +} + +#[derive(Clone)] +pub struct RedditBroker { + tx: mpsc::Sender, +} + +#[derive(Clone)] +struct RedditCredentials { + client_id: String, + client_secret: String, +} + +struct OAuthToken { + access_token: String, + expires_at: Instant, +} + +impl RedditBroker { + pub fn spawn(tree: Arc>, user_agent: &str) -> Self { + let (tx, rx) = mpsc::channel(100); + + let mut headers = header::HeaderMap::new(); + headers.insert( + header::USER_AGENT, + header::HeaderValue::from_str(user_agent).expect("valid user agent"), + ); + + let client = Client::builder() + .default_headers(headers) + .timeout(Duration::from_secs(15)) + .build() + .expect("reqwest client"); + + let creds = RedditCredentials::from_env(); + tokio::spawn(reddit_worker(rx, tree, client, creds)); + + Self { tx } + } + + /// Fire-and-forget: queue a fetch; worker updates the tree when done. + pub fn request_fetch(&self, id: ItemId) { + let _ = self.tx.try_send(RedditCommand { id }); + } +} + +impl RedditCredentials { + fn from_env() -> Option { + let client_id = std::env::var("REDDIT_CLIENT_ID").ok()?; + let client_secret = std::env::var("REDDIT_CLIENT_SECRET").ok()?; + if client_id.is_empty() || client_secret.is_empty() { + return None; + } + Some(Self { + client_id, + client_secret, + }) + } +} + +pub fn default_user_agent() -> String { + std::env::var("REDDIT_USER_AGENT").unwrap_or_else(|_| { + "web:sorter2.social:v0.0.1 (by /u/sorter2)".to_string() + }) } -/// Apply fetched entity data to a node (called from async worker). -pub fn apply_entity(tree: &mut GlobalTree, id: &ItemId, data: EntityData) { - tree.set_entity_data(id, data); +async fn reddit_worker( + mut rx: mpsc::Receiver, + tree: Arc>, + client: Client, + creds: Option, +) { + let mut in_flight = HashSet::new(); + let mut recently_fetched: HashMap = HashMap::new(); + let mut current_delay = Duration::from_secs(1); + let mut oauth: Option = None; + let cache_ttl = Duration::from_secs(300); + + while let Some(cmd) = rx.recv().await { + let now = Instant::now(); + recently_fetched.retain(|_, t| now.duration_since(*t) < cache_ttl); + + if in_flight.contains(&cmd.id) || recently_fetched.contains_key(&cmd.id) { + continue; + } + + in_flight.insert(cmd.id.clone()); + let fetch_id = cmd.id.clone(); + + tokio::time::sleep(current_delay).await; + + if let Some(c) = &creds { + oauth = ensure_oauth_token(&client, c, oauth.take()).await; + } + + let token = oauth.as_ref().map(|t| t.access_token.as_str()); + let use_oauth = token.is_some(); + + match do_fetch(&client, &fetch_id, use_oauth, token).await { + Ok(FetchOutcome::Entity(data)) => { + let mut w = tree.write().await; + w.set_entity_data(&fetch_id, data); + recently_fetched.insert(fetch_id.clone(), Instant::now()); + current_delay = Duration::from_millis(600); + } + Ok(FetchOutcome::NotFound) => { + recently_fetched.insert(fetch_id.clone(), Instant::now()); + } + Ok(FetchOutcome::RateLimited { reset_secs }) => { + let wait = Duration::from_secs(reset_secs.max(1)); + tracing::warn!( + "Reddit rate limit for {}; sleeping {}s", + fetch_id, + wait.as_secs() + ); + tokio::time::sleep(wait).await; + current_delay = (current_delay * 2).min(Duration::from_secs(60)); + } + Err(e) => { + tracing::warn!("Reddit fetch failed for {}: {}", fetch_id, e); + current_delay = (current_delay * 2).min(Duration::from_secs(60)); + } + } + + in_flight.remove(&fetch_id); + } +} + +enum FetchOutcome { + Entity(EntityData), + NotFound, + RateLimited { reset_secs: u64 }, +} + +async fn ensure_oauth_token( + client: &Client, + creds: &RedditCredentials, + existing: Option, +) -> Option { + if let Some(t) = existing { + if Instant::now() < t.expires_at - Duration::from_secs(60) { + return Some(t); + } + } + + let resp = client + .post("https://www.reddit.com/api/v1/access_token") + .basic_auth(&creds.client_id, Some(&creds.client_secret)) + .form(&[("grant_type", "client_credentials")]) + .send() + .await; + + let resp = match resp { + Ok(r) => r, + Err(e) => { + tracing::warn!("Reddit OAuth token request failed: {e}"); + return None; + } + }; + + if !resp.status().is_success() { + tracing::warn!("Reddit OAuth token HTTP {}", resp.status()); + return None; + } + + #[derive(Deserialize)] + struct TokenResponse { + access_token: String, + expires_in: u64, + } + + let body: TokenResponse = match resp.json().await { + Ok(b) => b, + Err(e) => { + tracing::warn!("Reddit OAuth token parse failed: {e}"); + return None; + } + }; + + Some(OAuthToken { + access_token: body.access_token, + expires_at: Instant::now() + Duration::from_secs(body.expires_in), + }) +} + +async fn do_fetch( + client: &Client, + id: &ItemId, + use_oauth: bool, + bearer: Option<&str>, +) -> Result { + let url = map_item_to_reddit_api(id, use_oauth); + if url.is_empty() { + return Ok(FetchOutcome::NotFound); + } + + let mut req = client.get(&url); + if let Some(token) = bearer { + req = req.bearer_auth(token); + } + + let resp = req.send().await.map_err(|e| e.to_string())?; + + if resp.status() == StatusCode::TOO_MANY_REQUESTS { + let reset = rate_limit_reset_secs(&resp); + return Ok(FetchOutcome::RateLimited { reset_secs: reset }); + } + + if resp.status() == StatusCode::SERVICE_UNAVAILABLE { + return Err("Reddit unavailable (503)".to_string()); + } + + if !resp.status().is_success() { + return Ok(FetchOutcome::NotFound); + } + + if rate_limit_remaining(&resp) == Some(0) { + let reset = rate_limit_reset_secs(&resp); + return Ok(FetchOutcome::RateLimited { reset_secs: reset }); + } + + let bytes = resp.bytes().await.map_err(|e| e.to_string())?; + Ok(parse_reddit_json(id, &bytes) + .map(FetchOutcome::Entity) + .unwrap_or(FetchOutcome::NotFound)) +} + +fn rate_limit_remaining(resp: &reqwest::Response) -> Option { + resp.headers() + .get("x-ratelimit-remaining") + .and_then(|v| v.to_str().ok()) + .and_then(|s| s.parse::().ok()) + .map(|f| f.floor() as u64) +} + +fn rate_limit_reset_secs(resp: &reqwest::Response) -> u64 { + resp.headers() + .get("x-ratelimit-reset") + .and_then(|v| v.to_str().ok()) + .and_then(|s| s.parse::().ok()) + .map(|f| f.ceil() as u64) + .unwrap_or(5) +} + +/// Map canonical item id to Reddit JSON API URL. +pub fn map_item_to_reddit_api(id: &ItemId, oauth: bool) -> String { + let path = id.as_str(); + if !path.starts_with("reddit.com/") && path != "reddit.com" { + return String::new(); + } + + let base = if oauth { + "https://oauth.reddit.com" + } else { + "https://www.reddit.com" + }; + + let segments: Vec<&str> = path.split('/').collect(); + + if let Some(i) = segments.iter().position(|&p| p == "comments") { + if segments.len() > i + 1 { + let api_path = segments[1..=i + 1].join("/"); + return format!("{base}/{api_path}.json?raw_json=1"); + } + } + + if segments.len() == 3 && segments[1] == "r" { + return format!("{base}/r/{}/about.json?raw_json=1", segments[2]); + } + + String::new() +} + +fn parse_reddit_json(id: &ItemId, bytes: &[u8]) -> Option { + let v: serde_json::Value = serde_json::from_slice(bytes).ok()?; + let segments: Vec<&str> = id.as_str().split('/').collect(); + + if segments.iter().any(|&p| p == "comments") { + parse_post_listing(&v) + } else { + parse_subreddit_about(&v) + } +} + +fn parse_subreddit_about(v: &serde_json::Value) -> Option { + let data = v.get("data")?; + let title = data + .get("title") + .or_else(|| data.get("display_name")) + .and_then(|t| t.as_str())? + .to_string(); + let body_html = data + .get("public_description_html") + .or_else(|| data.get("public_description")) + .and_then(|t| t.as_str()) + .map(|s| s.to_string()); + let thumb_url = data + .get("icon_img") + .or_else(|| data.get("community_icon")) + .and_then(|t| t.as_str()) + .filter(|s| !s.is_empty()) + .map(|s| s.to_string()); + + Some(EntityData { + title, + author: None, + body_html, + thumb_url, + }) +} + +fn parse_post_listing(v: &serde_json::Value) -> Option { + let listing = v.as_array()?.first()?; + let child = listing + .pointer("/data/children/0/data")?; + let title = child.get("title")?.as_str()?.to_string(); + let author = child + .get("author") + .and_then(|a| a.as_str()) + .filter(|a| *a != "[deleted]") + .map(|s| s.to_string()); + let body_html = child + .get("selftext_html") + .and_then(|t| t.as_str()) + .filter(|s| !s.is_empty()) + .map(|s| s.to_string()); + let thumb_url = child + .get("thumbnail") + .and_then(|t| t.as_str()) + .filter(|s| s.starts_with("http")) + .map(|s| s.to_string()); + + Some(EntityData { + title, + author, + body_html, + thumb_url, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn map_subreddit_about_url() { + let id = ItemId::parse("reddit.com/r/rust").unwrap(); + assert_eq!( + map_item_to_reddit_api(&id, false), + "https://www.reddit.com/r/rust/about.json?raw_json=1" + ); + assert_eq!( + map_item_to_reddit_api(&id, true), + "https://oauth.reddit.com/r/rust/about.json?raw_json=1" + ); + } + + #[test] + fn map_post_url() { + let id = + ItemId::parse("reddit.com/r/amitheasshole/comments/1trnvdl").unwrap(); + assert_eq!( + map_item_to_reddit_api(&id, false), + "https://www.reddit.com/r/amitheasshole/comments/1trnvdl.json?raw_json=1" + ); + } + + #[test] + fn map_non_reddit_empty() { + let id = ItemId::opaque("example.com/foo"); + assert!(map_item_to_reddit_api(&id, false).is_empty()); + } + + #[test] + fn parse_subreddit_fixture() { + let json = r#"{"kind":"t5","data":{"title":"Rust","display_name":"rust","public_description":"systems"}}"#; + let entity = parse_reddit_json( + &ItemId::parse("reddit.com/r/rust").unwrap(), + json.as_bytes(), + ) + .unwrap(); + assert_eq!(entity.title, "Rust"); + } + + #[test] + fn parse_post_fixture() { + let json = r#"[{"kind":"Listing","data":{"children":[{"kind":"t3","data":{"title":"AITA","author":"op","selftext_html":"<p>hi</p>","thumbnail":"https://b.thumbs.redditmedia.com/x.jpg"}}]}}]"#; + let entity = parse_reddit_json( + &ItemId::parse("reddit.com/r/x/comments/abc").unwrap(), + json.as_bytes(), + ) + .unwrap(); + assert_eq!(entity.title, "AITA"); + assert_eq!(entity.author.as_deref(), Some("op")); + } } diff --git a/server/src/state.rs b/server/src/state.rs index cc1722f5a5bf4d415f2327ea585c488a15a75592..8c03aa60c15aee803a534439400b69935b1a3d84 100644 --- a/server/src/state.rs +++ b/server/src/state.rs @@ -5,9 +5,10 @@ use tokio::sync::RwLock; use crate::{ event_log::EventLog, events::Event, + journal::JournalClient, path_types::ItemId, + reddit::{default_user_agent, RedditBroker}, reducer::{GlobalTree, VoteData}, - journal::JournalClient, views::ViewStore, }; @@ -73,6 +74,7 @@ pub struct AppState { pub views: ViewStore, pub tree: Arc>, journal: JournalClient, + pub reddit: RedditBroker, } impl AppState { @@ -112,6 +114,7 @@ impl AppState { let tree = Arc::new(RwLock::new(tree)); let journal = JournalClient::spawn(tree.clone(), event_log.clone()); + let reddit = RedditBroker::spawn(tree.clone(), &default_user_agent()); Self { cfg: Arc::new(cfg), @@ -119,6 +122,7 @@ impl AppState { views, tree, journal, + reddit, } } @@ -127,8 +131,11 @@ impl AppState { id: id.as_str().to_string(), }; self.event_log.append(&event).await.map_err(|e| e.to_string())?; - let mut w = self.tree.write().await; - w.ensure_path(id); + { + let mut w = self.tree.write().await; + w.ensure_path(id); + } + self.reddit.request_fetch(id.clone()); Ok(()) } diff --git a/todo b/todo new file mode 100644 index 0000000000000000000000000000000000000000..d196e8cb4cc80ccb95eeff01c73607d520e13212 --- /dev/null +++ b/todo @@ -0,0 +1,7 @@ +reddit import (only on explicit request) +reddit rendering +vote redering +pair chosing +nsfw gate + +logins (uuid user, two sides, oauths, and pseudonyms)