{"id":"W4408440538","doi":"10.5194/egusphere-egu25-4596","title":"ContrailBench: evaluating the performance of contrail models","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Environmental science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009798397,0.004448466,0.001304843,0.003518359,0.001005128,0.00239738,0.004184262,0.003621477,0.002080958],"category_scores_gemma":[0.01385377,0.0007265564,0.002445618,0.001504999,0.001279516,0.003503344,0.002026415,0.002718717,0.001402111],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002431988,"about_ca_system_score_gemma":0.001859165,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04460884,"about_ca_topic_score_gemma":0.03461586,"domain_scores_codex":[0.9960751,0.001086845,0.0004764555,0.001312725,0.0007619431,0.0002868117],"domain_scores_gemma":[0.9888931,0.00629044,0.000961274,0.001790547,0.001526985,0.0005377089],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001133747,0.001008898,0.05680461,0.0005936271,0.00120968,0.0001868963,0.0001568251,0.8609682,0.002420754,0.001377294,0.02118465,0.05295486],"study_design_scores_gemma":[0.00008290666,0.0003738949,0.004653559,0.00003927187,0.00006285258,0.00006556955,0.00007718403,0.9887952,0.002607709,0.0009594246,0.002233596,0.00004882381],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8417223,0.006087169,0.06076918,0.002702109,0.001419037,0.0009766169,0.04464331,0.03186577,0.009814533],"genre_scores_gemma":[0.7970989,0.0008418988,0.1010249,0.001171218,0.0003533552,0.0004263196,0.09511131,0.001166577,0.002805634],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04460884,"threshold_uncertainty_score":0.08869839,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5004050430154415,"score_gpt":0.5058007409941256,"score_spread":0.005395697978684111,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}