Side B implements a substantive behavioral change (per-commit rather than per-contributor ranking, with score rollups and UI updates) backed by new/updated tests across three test files, representing meaningful lasting functionality. Side A is a small, correct bugfix (reordering a guard check) with a single test update, which is valuable but far more limited in scope and impact than B's structural rework.
constitution · epochs · watch · epoch 3
c_abd68b5e771a (tommy-mor) vs c_fbeec5c4ad18 (tommy-mor)
download prompt · raw event · cmp_7693a31cebcef7
council reasoning
B redesigns ranking from contributor-bundled short-circuiting to pairwise ranking of every eligible commit with score rollup, evidence/UI surfaces, and tests—core lasting protocol behavior. A is a correct but narrow bugfix that only reorders the zero-ratio guard before ensure_item/voted_pairs side effects.
Side B changes the core ranking behavior from contributor-level to commit-level by comparing every eligible commit, rolling scores back up to contributors, updating prompts, evidence, UI, and adding tests for same-contributor multi-commit cases and new ranking outputs. Side A fixes a real reducer bug by moving the zero-ratio guard before side effects to prevent ghost items and voted pairs, with an accompanying regression test, but its scope is much narrower than the architectural change in Side B.
sides
A — c_abd68b5e771a (tommy-mor)
message
[81de487b] Fix zero-ratio guard in reducer to drop before registering items or pair. Previously the early-return for zero-weight votes happened after ensure_item and voted_pairs.insert, leaving ghost items in the index and the pair incorrectly marked as voted. Move the check to before any side effects. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
diff preview
diff --git a/server/src/reducer.rs b/server/src/reducer.rs
index 6841d35cfc9de2389f340a22b8a45acb335e36c3..0e36979abe0f051493038ff7e652efc7f7a0ac80 100644
--- a/server/src/reducer.rs
+++ b/server/src/reducer.rs
@@ -112,6 +112,10 @@ impl GroupState {
if vote.ratio_right < 0 {
vote.ratio_right = 0;
}
+ if vote.ratio_left == 0 || vote.ratio_right == 0 {
+ // Zero on either side produces no valid edge; drop before registering items or pair.
+ return;
+ }
let a_idx = self.ensure_item(&vote.a);
let b_idx = self.ensure_item(&vote.b);
@@ -121,10 +125,6 @@ impl GroupState {
let w_a = vote.ratio_left as f64;
let w_b = vote.ratio_right as f64;
- if w_a == 0.0 || w_b == 0.0 {
- // Zero on either side produces no valid edge; drop the vote.
- return;
- }
self.add_edge_weight(b_idx, a_idx, w_a);
self.add_edge_weight(a_idx, b_idx, w_b);
diff --git a/server/tests/basic.rs b/server/tests/basic.rs
index cc8c1a0d139f3722ba6ecd13dd001c65be835b67..08159f4a7f0850fd165817a4a1af4f31ced2ad76 100644
--- a/server/tests/basic.rs
+++ b/server/tests/basic.rs
@@ -546,12 +546,10 @@ fn reducer_negative_ratio_clamped_to_zero() {
delegate: Some("00000000-0000-0000-0000-000000000000:test:local/test".to_string()),
thread_tag: "t".to_string(),
});
- // Items are registered, but the zero-clamped vote produces no edges.
- assert_eq!(group.idx_to_item.len(), 2);
- let a_idx = group.item_to_idx[&item_id("https://slug.social/~/t/a")];
- let b_idx = group.item_to_idx[&item_id("https://slug.social/~/t/b")];
- assert!(!group.edges.contains_key(&(a_idx, b_idx)));
- assert!(!group.edges.contains_key(&(b_idx, a_idx)));
+ // Nothing registered: zero-clamped vote is dropped before ensure_item.
+ assert!(group.idx_to_item.is_empty());
+ assert!(group.edges.is_empty());
+ assert!(group.voted_pairs.is_empty());
}
B — c_fbeec5c4ad18 (tommy-mor)
message
[c7ef287e] Rank every eligible commit with the LLM council. Stop short-circuiting on a single contributor; pairwise-sort commits, roll scores up for emission payouts, and surface commit rankings on epoch pages. Co-authored-by: Cursor <cursoragent@cursor.com>
diff preview
diff --git a/constitution.py b/constitution.py
index 26ba130e885e0e69fb7874ca5c3f07f42100a150..71fc46b4c9a4d7a9ea7bf319860b0db1acc4a673 100644
--- a/constitution.py
+++ b/constitution.py
@@ -503,20 +503,22 @@ def _epochs_in_ledger() -> list[int]:
def build_pairwise_prompt(side_a: dict, side_b: dict) -> str:
- return f"""You are ranking contributions to an open source project.
-Compare these two sides (each may be one or more commits). Decide which side contributed more.
+ return f"""You are ranking individual git commits to an open source project.
+Compare these two commits. Decide which commit contributed more.
Return ONLY a JSON object: {{"winner": "A" or "B", "ratio": "N:M", "explanation": "..."}}
-Side A — commit messages:
+Side A — contributor: {side_a.get('contributor', '?')}
+Side A — commit message:
{side_a['message']}
-Side A — unified diffs (full patches):
+Side A — unified diff (full patch):
{side_a['diff']}
-Side B — commit messages:
+Side B — contributor: {side_b.get('contributor', '?')}
+Side B — commit message:
{side_b['message']}
-Side B — unified diffs (full patches):
+Side B — unified diff (full patch):
{side_b['diff']}"""
@@ -1518,16 +1520,28 @@ async def broadcast_js(js: str):
await queue.put(js)
-def _author_side_for_llm(author: str, author_commits: dict) -> dict:
- cs = author_commits[author]
+def _commit_side_for_llm(row: dict) -> dict:
+ oid = row["oid"]
+ short = oid.split(":", 1)[1][:8] if ":" in oid else oid[:8]
return {
- "message": "\n".join(f"[{c['sha']}] {c['message']}" for c in cs),
- "diff": "\n\n".join(f"=== {c['sha']} ===\n{c['diff']}" for c in cs),
- "commit_ids": [c["commit_id"] for c in cs],
- "contributor": author,
+ "message": f"[{short}] {row['message']}",
+ "diff": row["patch"] or "",
+ "commit_id": commit_id_for_oid(oid),
+ "contributor": row["contributor"],
+ "oid": oid,
}
+def _rollup_contributor_scores(
+ ordered: list[dict], commit_scores: list[Decimal]
+) -> dict[str, Decimal]:
+ totals: dict[str, Decimal] = {}
+ for row, score in zip(ordered, commit_scores):
+ contributor = row["contributor"]
+ totals[contributor] = totals.get(contributor, Decimal("0")) + score
+ return totals
+
+
def _find_judgment(comparison_id: str, model_id: str) -> dict | None:
for e in evidence_by_kind("llm.judgment"):
p = e.payload
@@ -1546,101 +1560,106 @@ def _find_ranking_models(ranking_run_id: str) -> list[str] | None:
async def rank_commits(commits: list[dict], *, epoch: int = -1):
+ """Pairwise-rank every eligible commit; roll scores up to contributors."""
if not commits:
return {}, [], {"ranking_run_id": "", "ranking_event_id": ""}
- commit_ids = sorted(commit_id_for_oid(row["oid"]) for row in commits)
+ ordered = sorted(commits, key=lambda r: r["oid"])
+ commit_ids = [commit_id_for_oid(row["oid"]) for row in ordered]
ranking_run_id = _content_id("rank", {
"epoch": epoch,
- "commit_ids": commit_ids,
+ "commit_ids": sorted(commit_ids),
})
- contributors = sorted(set(c["contributor"] for c in commits))
+ contributors = sorted({c["contributor"] for c in ordered})
- if len(contributors) == 1:
+ # Nothing to compare: a single commit (not a single contributor).
+ if len(ordered) == 1:
await append_evidence(epoch, "ranking.started", {
"ranking_run_id": ranking_run_id,
"commit_ids": commit_ids,
"contributors": contributors,
"models": [],
- "summary": f"ranking epoch {epoch}: single contributor",
+ "summary": f"ranking epoch {epoch}: single commit",
})
- ranking = {contributors[0]: Decimal("1")}
+ commit_ranking = {commit_ids[0]: "1"}
+ contributor_ranking = {ordered[0]["contributor"]: Decimal("1")}
completed = await append_evidence(epoch, "ranking.completed", {
"ranking_run_id": ranking_run_id,
"models": [],
- "ranking": {contributors[0]: "1"},
+ "commit_ranking": commit_ranking,
+ "contributor_ranking": {ordered[0]["contributor"]: "1"},
+ "ranking": {ordered[0]["contributor"]: "1"},
"judgment_ids": [],
- "summary": f"Only {contributors[0]} is eligible; rank is 1.0",
+ "summary": f"Only one eligible commit; {ordered[0]['contributor']} rank 1.0",
})
await broadcast_audit(
"ranking",
- f"Only {contributors[0]} is eligible; rank is 1.0",
+ f"Only one eligible commit; {ordered[0]['contributor']} rank 1.0",
progress=90,
phase="finalizing",
evidence_event_id=completed.event_id,
evidence_url=_evidence_url("event", completed.event_id),
links={"epoch": _evidence_url("epoch", str(epoch))},
)
- return ranking, [], {
+ return contributor_ranking, [], {
"ranking_run_id": ranking_run_id,
"ranking_event_id": completed.event_id,
}
if not (OPENROUTER_API_KEY or "").strip():
raise RuntimeError(
- "OPENROUTER_API_KEY is required when multiple contributors need ranking"
+ "OPENROUTER_API_KEY is required when multiple commits need ranking"
)
models = _find_ranking_models(ranking_run_id)
if models is None:
models = await fetch_top_models(n=3)
if not models:
- raise RuntimeError("no council models available for contributor ranking")
+ raise RuntimeError("no council models available for commit ranking")
await append_evidence(epoch, "ranking.started", {
"ranking_run_id": ranking_run_id,
"commit_ids": commit_ids,
"contributors": contributors,
"models": models,
- "summary": f"Council selected: {', '.join(models)}",
+ "summary": (
+ f"Council selected: {', '.join(models)} — "
+ f"{len(ordered)} commits"
+ ),
})
await broadcast_audit(
"council",
- f"Council selected: {', '.join(models)}",
+ f"Council selected: {', '.join(models)} — ranking {len(ordered)} commits",
progress=35,
phase="ranking",
)
await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][
- ["div.log-council", f"Council: {', '.join(models)} — {len(commits)} commits"]
+ ["div.log-council",
+ f"Council: {', '.join(models)} — {len(ordered)} commits"]
]))
- authors = contributors
- author_commits = {a: [] for a in authors}
- for row in sorted(commits, key=lambda r: r["oid"]):
- author_commits[row["contributor"]].append({
- "message": row["message"],
- "sha": row["oid"].split(":", 1)[1][:8],
- "diff": row["patch"],
- "commit_id": commit_id_for_oid(row["oid"]),
- })
-
+ sides = [_commit_side_for_llm(row) for row in ordered]
judgment_ids: list[str] = []
async def compare_fn(i, j):
- a1, a2 = authors[i], authors[j]
- side_a = _author_side_for_llm(a1, author_commits)
- side_b = _author_side_for_llm(a2, author_commits)
+ side_a, side_b = sides[i], sides[j]
+ label_a = f"{side_a['commit_id'][:16]} ({side_a['contributor']})"
+ label_b = f"{side_b['commit_id'][:16]} ({side_b['contributor']})"
prompt = build_pairwise_prompt(side_a, side_b)
comparison_material = {
"ranking_run_id": ranking_run_id,
"side_a": {
- "contributor": a1,
- "commit_ids": side_a["commit_ids"],
+ "contributor": side_a["contributor"],
+ "commit_id": side_a["commit_id"],
+ "commit_ids": [side_a["commit_id"]],
+ "oid": side_a["oid"],
"message": _bytes_blob(side_a["message"]),
"diff": _bytes_blob(side_a["diff"]),
},
"side_b": {
- "contributor": a2,
- "commit_ids": side_b["commit_ids"],
+ "contributor": side_b["contributor"],
+ "commit_id": side_b["commit_id"],
+ "commit_ids": [side_b["commit_id"]],
+ "oid": side_b["oid"],
"message": _bytes_blob(side_b["message"]),
"diff": _bytes_blob(side_b["diff"]),
},
@@ -1650,22 +1669,24 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1):
comparison_material = {
**comparison_material,
"comparison_id": comparison_id,
- "summary": f"Comparing {a1} with {a2}",
+ "summary": f"Comparing {label_a} with {label_b}",
}
cmp_ev = await append_evidence(epoch, "comparison.input", comparison_material)
await broadcast_audit(
"comparison",
- f"Comparing {a1} with {a2}",
+ f"Comparing commits {label_a} vs {label_b}",
phase="ranking",
evidence_event_id=cmp_ev.event_id,
evidence_url=_evidence_url("comparison", comparison_id),
links={
"comparison": _evidence_url("comparison", comparison_id),
+ "commit_a": _evidence_url("commit", side_a["commit_id"]),
+ "commit_b": _evidence_url("commit", side_b["commit_id"]),
"epoch": _evidence_url("epoch", str(epoch)),
},
)
await broadcast_js(exec_event(Three[Selector("#emission-status")][MORPH][
- ["div#emission-status", f"Comparing {a1} vs {a2}…"]
+ ["div#emission-status", f"Comparing {label_a} vs {label_b}…"]
]))
results = []
for model in models:
@@ -1697,9 +1718,15 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1):
jud_id = (existing or _find_judgment(comparison_id, model) or {}).get(
"judgment_id"
)
+ win_label = (
+ f"{sides[w]['commit_id'][:16]} ({sides[w]['contributor']})"
+ )
+ lose_label = (
+ f"{sides[l]['commit_id'][:16]} ({sides[l]['contributor']})"
+ )
await broadcast_audit(
"vote",
- f"{model}: {authors[w]} over {authors[l]} ({result['ratio']})",
+ f"{model}: {win_label} over {lose_label} ({result['ratio']})",
phase="ranking",
evidence_url=(
_evidence_url("judgment", jud_id) if jud_id else None
@@ -1714,8 +1741,8 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1):
await broadcast_js(exec_event(Three[Selector("#emission-log")][PREPEND][
["div.log-vote",
["span.model", model], " — ",
- ["span.winner", authors[w]], f" beat ",
- ["span.loser", authors[l]], f" ({result['ratio']}) ",
+ ["span.winner", win_label], f" beat ",
+ ["span.loser", lose_label], f" ({result['ratio']}) ",
["span.explanation", result["explanation"]],
]
]))
@@ -1745,24 +1772,46 @@ async def rank_commits(commits: list[dict], *, epoch: int = -1):
["div#emission-status", label]
]))
- pairs = await pairwise_rank(len(authors), compare_fn, progress_fn)
+ pairs = await pairwise_rank(len(ordered), compare_fn, progress_fn)
if not pairs:
- ranking = {authors[0]: Decimal("1")} if authors else {}
+ commit_score_list = [Decimal("1")]
else:
scores = rank_centrality(pairs)
- ranking = {authors[i]: Decimal(str(scores[i])) for i in range(len(authors))}
… preview truncated; 12,453 characters omittedHardlinks — judgments / attempts / prompt
judgments
attempts
Prompt text is loaded only by the download route.