{"id":"W4409361388","doi":"10.1609/aaai.v39i26.34944","title":"Joint Scoring Rules: Competition Between Agents Avoids Performative Prediction","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Multi-Agent Systems and Negotiation","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Performative utterance; Joint (building); Competition (biology); Computer science; Artificial intelligence; Epistemology; Engineering; Philosophy; Structural engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007872964,0.0001963242,0.0002615062,0.0002336798,0.0004076776,0.0002570338,0.00102286,0.00009521162,0.00003288553],"category_scores_gemma":[0.0002328187,0.0001547121,0.0001142813,0.0005609462,0.00009492074,0.000733387,0.0003070649,0.0003213444,0.0001004948],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001203996,"about_ca_system_score_gemma":0.00007989019,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005257577,"about_ca_topic_score_gemma":0.000002969781,"domain_scores_codex":[0.9981155,0.00004104021,0.0007065561,0.0004050879,0.0004716974,0.0002601047],"domain_scores_gemma":[0.9986222,0.00007031148,0.0004385225,0.0002574925,0.000551427,0.0000600156],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00001774035,0.0001007434,0.004159262,0.0001606142,0.00003572637,1.198342e-7,0.003262285,0.00009347194,0.02256794,0.9352877,0.0002031815,0.03411118],"study_design_scores_gemma":[0.00007780084,0.0001725245,0.02447568,0.001594092,0.00002981281,0.000001444401,0.001212793,0.2227561,0.6750085,0.0742485,0.0001753193,0.0002473539],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6774993,0.00001585303,0.2993834,0.002005434,0.001285868,0.0008100488,0.0000165692,0.0001751248,0.01880844],"genre_scores_gemma":[0.9982923,0.00002570001,0.001223169,0.00009727373,0.00009273754,0.00003306007,0.000002310978,0.000006201939,0.0002271969],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8610392,"threshold_uncertainty_score":0.6308975,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.107909667531006,"score_gpt":0.3036342043271588,"score_spread":0.1957245367961528,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}