Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion delphi/docs/CUTOVER_RUNBOOK.md
Original file line number Diff line number Diff line change
Expand Up @@ -63,7 +63,7 @@ from Clojure's env, rows invisible to the server (UNIQUE(zid, math_env)).

```
docker compose --profile delphi-math up -d delphi-math-poller
# env: POLISMATH_ENGINE_MODE=clojure-legacy DELPHI_MATH_ENV=delphi
# env: DELPHI_MATH_ENV=delphi (engine has one path since the mode collapse)
```

Verify within minutes:
Expand Down
41 changes: 40 additions & 1 deletion delphi/polismath/pca_kmeans_rep/repness.py
Original file line number Diff line number Diff line change
Expand Up @@ -183,7 +183,14 @@ def compute_group_comment_stats_df(votes_long: pd.DataFrame,

Vectorized port of Clojure's per-(group, comment) `comment-stats` recipe
(math/src/polismath/math/repness.clj:64-100). Operates on all groups and
comments simultaneously.
comments simultaneously, in two phases:

1. :func:`_group_comment_vote_counts` — the DataFrame plumbing that
reduces (votes, group memberships) to one row of raw counts per
(group, comment): ``na``/``nd``/``ns`` for the group and
``other_agree``/``other_disagree``/``other_votes`` for everyone else.
2. :func:`_comment_stats_from_counts` — the statistics recipe, reading
like Clojure's scalar comment-stats/finalize-cmt-stats chain.

Args:
votes_long: Long-format DataFrame with columns:
Expand All @@ -210,6 +217,25 @@ def compute_group_comment_stats_df(votes_long: pd.DataFrame,
- disagree_metric: metric for disagree representativeness
- repful: 'agree' or 'disagree' based on which is more representative
"""
counts_df = _group_comment_vote_counts(votes_long, group_clusters, tid_order)
if counts_df.empty:
return counts_df
return _comment_stats_from_counts(counts_df)


def _group_comment_vote_counts(votes_long: pd.DataFrame,
group_clusters: List[Dict[str, Any]],
tid_order: Optional[List[Any]] = None) -> pd.DataFrame:
"""Phase 1 — the plumbing: reduce (votes, group memberships) to raw
per-(group, comment) counts.

Returns a DataFrame indexed by (group_id, comment) with the group's
``na``/``nd``/``ns``, the clustered-voter totals ``total_agree``/
``total_disagree``/``total_votes``, and the derived ``other_*`` columns
(everyone not in this group) — the exact inputs Clojure's comment-stats
recipe consumes. Empty result (correct schema) when there are no votes
or no clustered voters.
"""
# Build participant -> group mapping
ptpt_to_group = {}
for group in group_clusters:
Expand Down Expand Up @@ -304,6 +330,19 @@ def compute_group_comment_stats_df(votes_long: pd.DataFrame,
stats_df['other_disagree'] = stats_df['total_disagree'] - stats_df['nd']
stats_df['other_votes'] = stats_df['total_votes'] - stats_df['ns']

return stats_df


def _comment_stats_from_counts(stats_df: pd.DataFrame) -> pd.DataFrame:
"""Phase 2 — the statistics recipe, on clean per-(group, comment) counts.

Reads like Clojure's scalar chain (comment-stats -> add-comparative-stats
-> finalize-cmt-stats, repness.clj:64-100/:97-100/:178/:191-193):
probabilities with pseudocounts, proportion tests on raw counts,
representativeness ratios group-vs-other, two-proportion tests, the
signed metric products, and the repful side pick. Adds the stat columns
to ``stats_df`` (same frame, mutated in place) and returns it.
"""
# Compute probabilities with pseudocounts (Bayesian smoothing)
# For group
stats_df['pa'] = (stats_df['na'] + PSEUDO_COUNT/2) / (stats_df['ns'] + PSEUDO_COUNT)
Expand Down
4 changes: 2 additions & 2 deletions delphi/tests/test_in_conv_greedy_carry.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,7 @@
2. Legacy mode admits the top (15 - n) voters when under 15, with ties broken
by matrix row order (deterministic surrogate for Clojure's hash-order tie).
3. Legacy greedy admits PERSIST across ticks even once the conversation grows
past 15 threshold-qualifiers (the carry) — improved mode drops them.
past 15 threshold-qualifiers (the carry).
4. Threshold-qualifiers stay in across ticks in BOTH modes (monotonicity).
"""

Expand Down Expand Up @@ -114,7 +114,7 @@ def test_legacy_blob_in_conv_includes_greedy_admits(self, monkeypatch):

class TestThresholdMonotonicity:

@pytest.mark.parametrize('mode', ['improved', 'clojure-legacy'])
@pytest.mark.parametrize('mode', ['clojure-legacy'])
def test_qualifier_stays_in_across_ticks(self, monkeypatch, mode):
_mode(monkeypatch, mode)
conv = Conversation('mono').update_votes(_votes(_TICK1_SPECS))
Expand Down
8 changes: 4 additions & 4 deletions delphi/tests/test_mod_ptpt_leak_parity.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,10 +8,10 @@
repness. Python gained a real ban feature (mod_out_ptpts row drop in
_apply_moderation, 2026-06-10) — correct, but a certification divergence.

Per Q1: 'clojure-legacy' mode must LEAK the ban exactly like Clojure (rows
kept everywhere); 'improved' mode keeps the fix. This supersedes the legacy-
mode premise of #2623's TestCarryPruneOnParticipantBan (banning can no longer
shrink the legacy clustering pool, so the carry-prune scenario cannot arise).
Per Q1 + the mode collapse (2026-07-27, bans dropped as a feature): the
engine LEAKS the ban exactly like Clojure (rows kept everywhere). This
supersedes #2623's TestCarryPruneOnParticipantBan (banning can no longer
shrink the clustering pool, so the carry-prune scenario cannot arise).
"""

import os
Expand Down
76 changes: 75 additions & 1 deletion delphi/tests/test_repness_unit.py
Original file line number Diff line number Diff line change
Expand Up @@ -456,4 +456,78 @@ def test_other_votes_includes_other_group_pass(self):
assert g1['other_votes'] == 2, (
f"group 1 other_votes should include group-0 PASS; "
f"got {g1['other_votes']}"
)
)

class TestBlobInjectionStats:
"""PR 14b: inject the CLOJURE blob's group memberships + the dataset's
votes into the PRODUCTION stats path (compute_group_comment_stats_df)
and compare per-(gid, tid) values against the blob's repness entries —
the non-tautological pin of the vectorized formulas against the oracle
(HANDOFF_PR14_VECTORIZED_REFACTOR.md task 2).

The blob stores only the WINNING side's values; `repful-for` selects
which of our columns to compare (agree -> na/pa/pat/ra/rat, disagree ->
nd/pd/pdt/rd/rdt). `repness-test` is emitted ROUNDED (~7 significant
digits) by Clojure, hence its looser tolerance.
"""

def _stats_and_blob(self, ds_name):
import json
from polismath.regression import get_dataset_files
from common_utils import create_test_conversation

files = get_dataset_files(ds_name, blob_type='cold_start')
blob_path = files['math_blob']
if not os.path.exists(blob_path):
pytest.skip(f"math blob for {ds_name} unavailable")
with open(blob_path) as fh:
blob = json.load(fh)

conv = create_test_conversation(ds_name)
votes_long = conv.rating_mat.melt(
ignore_index=False, var_name='comment', value_name='vote'
).reset_index(names='participant')

# Blob group members are BASE-cluster ids; unfold through the
# blob's own base-clusters (NOT python's clustering — injection).
# create_test_conversation matrices carry STRING pids/tids; the blob
# carries ints — map on the way in (and look tids up as str below).
bc = blob['base-clusters']
bid_to_pids = dict(zip(bc['id'], bc['members']))
groups = [
{'id': g['id'],
'members': [str(pid) for bid in g['members']
for pid in bid_to_pids[bid]]}
for g in blob['group-clusters']
]
stats = compute_group_comment_stats_df(votes_long, groups)
return stats, blob

@pytest.mark.parametrize('ds_name', ['vw', 'biodiversity'])
def test_stats_match_blob_repness_entries(self, ds_name):
stats, blob = self._stats_and_blob(ds_name)
checked = 0
for gid_str, entries in blob['repness'].items():
gid = int(gid_str)
for e in entries:
tid = str(e['tid'])
if (gid, tid) not in stats.index:
pytest.fail(f"blob entry (gid={gid}, tid={tid}) missing "
f"from stats index")
row = stats.loc[(gid, tid)]
side = e['repful-for']
n_col, p_col, pt_col, r_col, rt_col = (
('na', 'pa', 'pat', 'ra', 'rat') if side == 'agree'
else ('nd', 'pd', 'pdt', 'rd', 'rdt'))
assert row[n_col] == e['n-success'], (gid, tid, side)
assert row['ns'] == e['n-trials'], (gid, tid, side)
assert np.isclose(row[p_col], e['p-success'],
rtol=1e-9, atol=1e-12), (gid, tid, 'p-success')
assert np.isclose(row[pt_col], e['p-test'],
rtol=1e-9, atol=1e-12), (gid, tid, 'p-test')
assert np.isclose(row[r_col], e['repness'],
rtol=1e-9, atol=1e-12), (gid, tid, 'repness')
assert np.isclose(row[rt_col], e['repness-test'],
rtol=2e-6, atol=1e-9), (gid, tid, 'repness-test')
checked += 1
assert checked >= 4, f"vacuous: only {checked} blob entries compared"
Loading