{"id":"W4408427063","doi":"10.5194/egusphere-egu25-11024","title":"Skill assessment of a multi-system ensemble of initialized 20-year predictions","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Environment and Climate Change Canada","funders":"","keywords":"Computer science; Statistics; Artificial intelligence; Econometrics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00233013,0.0005380766,0.0005690123,0.000501401,0.0003476466,0.0006977451,0.0005838439,0.0005202881,0.0008204835],"category_scores_gemma":[0.005424982,0.00027207,0.0005575712,0.0004075784,0.0002997281,0.0009109089,0.0007008627,0.0006460496,0.0001300507],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004792564,"about_ca_system_score_gemma":0.0007656198,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01439795,"about_ca_topic_score_gemma":0.0112047,"domain_scores_codex":[0.9996902,0.0001093655,0.00003419001,0.00007241802,0.00005343836,0.00004044536],"domain_scores_gemma":[0.9975666,0.00143254,0.0002239261,0.0002530271,0.0004075825,0.000116364],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002066153,0.00007276302,0.03706062,0.00002634051,0.0002074135,0.0001025229,0.00007221921,0.9458194,0.001154571,0.0005560346,0.0003898364,0.01433173],"study_design_scores_gemma":[0.00001046225,0.00004931353,0.01107407,0.000006737966,0.00002532826,0.00001027911,0.00002023335,0.987703,0.0006830039,0.0002462852,0.0001596365,0.00001161],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9662868,0.0001288626,0.03084577,0.0001226409,0.00003509786,0.00004711578,0.0006016165,0.0001953785,0.001736622],"genre_scores_gemma":[0.9916435,0.00005255178,0.007131707,0.00002303392,0.00001489248,0.00001997683,0.0008776747,0.000015815,0.0002207905],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01439795,"threshold_uncertainty_score":0.02862829,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2566147092626727,"score_gpt":0.5422345913921873,"score_spread":0.2856198821295146,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}