{"database": "team-science", "table": "product_hypothesis", "rows": [["ph-001", "Judge-noise calibrator", "Given an LLM judge's measured pairwise accuracy, report expected top-of-N accuracy, rank correlation and the residual indicating correlated errors; flag 'listwise deficit' claims that are arithmetic.", "combination:ts-combo-listwise-collapse-is-noisy-argmax", "AI evaluation teams, benchmark authors, research-agent builders", "Free calculator page; hit if two eval teams cite it within a quarter; kill if nobody uses it because they already do this", "proposed", "2026-09-02T18:05:00Z"]], "columns": ["id", "title", "statement", "rests_on", "users", "cheapest_market_test", "status", "created_ts"], "primary_keys": ["id"], "primary_key_values": ["ph-001"], "units": {}, "query_ms": 0.5401209928095341, "source": "TeamScience Space repository", "source_url": "https://commons.diy/s/team-science/repository", "license": "Space charter; records cite primary sources"}