{"id":"W3124232740","doi":"10.20944/preprints201801.0217.v1","title":"Evaluation of Analysis by Cross-Validation. Part I: Using Verification Metrics","year":2018,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Air Quality Monitoring and Forecasting","field":"Environmental Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"Environment and Climate Change Canada","funders":"","keywords":"Set (abstract data type); Error analysis; Variance (accounting); Moment (physics); Computer science; Statistics; Margin (machine learning); Algorithm; Mathematics; Applied mathematics; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.008308646,0.0002324942,0.0003614396,0.0001925836,0.0001767854,0.00003848136,0.0005017742,0.0002926194,0.003610324],"category_scores_gemma":[0.001611293,0.000260692,0.0002258788,0.001025642,0.0002667298,0.0001843205,0.00108191,0.0002986884,0.0004387611],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008434211,"about_ca_system_score_gemma":0.00007077813,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001158821,"about_ca_topic_score_gemma":0.000006085219,"domain_scores_codex":[0.9957963,0.0005403734,0.0007666637,0.0008926208,0.00176157,0.0002424929],"domain_scores_gemma":[0.9973401,0.00007976565,0.0009780258,0.001191517,0.0003188064,0.00009176997],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00000669892,0.00008245074,0.854336,0.00002406939,0.000250695,7.708238e-8,0.0003965165,0.1356629,0.007566518,0.000002161699,0.00003717786,0.001634749],"study_design_scores_gemma":[0.0001842988,0.00001059694,0.6816754,0.00004649736,0.00205323,4.154549e-7,0.00005217954,0.1766582,0.1378677,0.0006203524,0.000491755,0.0003392425],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9905303,0.00004982157,0.005595025,0.0000177429,0.0005030173,0.000372307,0.00005060925,0.00005490001,0.002826289],"genre_scores_gemma":[0.997904,0.00002151195,0.001415948,0.000005929457,0.0001495683,0.00004726604,0.0001837917,0.00002122506,0.0002507706],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1726605,"threshold_uncertainty_score":0.9999845,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3687192818685479,"score_gpt":0.4427413703578471,"score_spread":0.07402208848929925,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}