{"id":"W4410713958","doi":"10.31234/osf.io/mjx2v_v2","title":"Illusions of Confidence in Artificial Systems","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Computability, Logic, AI Algorithms","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council; UK Research and Innovation; HORIZON EUROPE Framework Programme; Government of the United Kingdom; Canadian Institute for Advanced Research","keywords":"Illusion; Artificial intelligence; Computer science; Cognitive psychology; Psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003267278,0.0003202546,0.0003159395,0.0009644401,0.0004854923,0.002757522,0.0005326576,0.001204325,0.001384833],"category_scores_gemma":[0.0457278,0.000358049,0.0002857933,0.0004009691,0.005214951,0.004945887,0.004566428,0.0023367,0.00009747744],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008511215,"about_ca_system_score_gemma":0.0001971664,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004035267,"about_ca_topic_score_gemma":0.0001767571,"domain_scores_codex":[0.9969304,0.001269532,0.0002134455,0.0004249121,0.0009845037,0.0001772555],"domain_scores_gemma":[0.9609993,0.02364387,0.006146685,0.006531233,0.001609746,0.001069073],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.003200996,0.0004002314,0.1772619,0.0009864479,0.000402497,0.001101641,0.0493749,0.02303205,0.1625476,0.3534726,0.002406209,0.2258129],"study_design_scores_gemma":[0.000123131,0.0007809258,0.2254364,0.0002476624,0.0001343238,0.001647585,0.004871665,0.06515285,0.03605625,0.6587633,0.006449619,0.000336297],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9412695,0.0005812067,0.04564187,0.001414214,0.00002728448,0.00002513332,0.00007485926,0.0002485853,0.01071738],"genre_scores_gemma":[0.997668,0.00004560848,0.002059163,0.00007838599,0.000007093125,0.000008760161,0.00001443216,0.00001183533,0.0001067167],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003267278,"threshold_uncertainty_score":0.01727921,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04108516131200808,"score_gpt":0.2953251531698814,"score_spread":0.2542399918578734,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}