{"id":"W7111079672","doi":"10.48550/arxiv.2512.05998","title":"Going All-In on LLM Accuracy: Fake Prediction Markets, Real Confidence Signals","year":2025,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Incentive; Framing (construction); Baseline (sea); Binary number; Scoring rule; Framing effect; Confidence interval; Task (project management)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009092703,0.0008268011,0.0006906939,0.0003337261,0.0005444109,0.002907213,0.0009065571,0.001524411,0.006640984],"category_scores_gemma":[0.07406826,0.0004891288,0.0005167942,0.0002660938,0.001235056,0.005379955,0.001740616,0.002396421,0.001285631],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001071112,"about_ca_system_score_gemma":0.0005900164,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002885914,"about_ca_topic_score_gemma":0.002476108,"domain_scores_codex":[0.9961569,0.001917512,0.0001977919,0.0008784407,0.000617965,0.0002313129],"domain_scores_gemma":[0.945857,0.03940923,0.004305824,0.007537153,0.001899386,0.0009914608],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01679778,0.00360927,0.3352302,0.0007413426,0.0007209654,0.0006499403,0.00485398,0.1546281,0.06255377,0.03412547,0.01266795,0.3734213],"study_design_scores_gemma":[0.0007569166,0.005828412,0.1820737,0.0001984358,0.0004213934,0.0005147842,0.001331026,0.7006079,0.04443255,0.0536702,0.00980824,0.000356433],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9802771,0.0001630095,0.01080887,0.001244828,0.00008844088,0.00004054938,0.0002301801,0.000261978,0.006885007],"genre_scores_gemma":[0.9956642,0.00002613002,0.003012871,0.0001860686,0.00002415406,0.00002229318,0.0001660653,0.00004258644,0.000855597],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009092703,"threshold_uncertainty_score":0.04808736,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1454807877301481,"score_gpt":0.2492995981184722,"score_spread":0.1038188103883241,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}