{"id":"W4389728957","doi":"10.21203/rs.3.rs-3736323/v1","title":"Prediction Performance Metrics Considering the Difficulty of Individual Cases","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Machine learning; Metric (unit); Performance prediction; Performance metric; Artificial intelligence; Predictive modelling; Artificial neural network; Variety (cybernetics); Data mining; Simulation; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01520447,0.002206256,0.001578757,0.004092757,0.0007591784,0.003174675,0.001523215,0.001926427,0.00115743],"category_scores_gemma":[0.0709624,0.0003762934,0.0009034059,0.002670499,0.00118302,0.005417776,0.002129944,0.002485236,0.0003951235],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001836225,"about_ca_system_score_gemma":0.001299141,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004028634,"about_ca_topic_score_gemma":0.002546717,"domain_scores_codex":[0.9881175,0.003735856,0.001454701,0.002187624,0.003954745,0.0005494765],"domain_scores_gemma":[0.8999283,0.06922241,0.009625979,0.007966074,0.01111692,0.002140295],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00103125,0.0008063841,0.1525086,0.0003852455,0.0005975093,0.0002375515,0.0003740382,0.668825,0.005622947,0.004187056,0.006891728,0.1585326],"study_design_scores_gemma":[0.00002002246,0.0002621956,0.01039962,0.00003124332,0.000042627,0.00007703735,0.00008281692,0.9800823,0.00426818,0.004188966,0.0005012933,0.00004364678],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7480231,0.00225102,0.2361378,0.001522916,0.0003679918,0.0004617844,0.001593101,0.002124538,0.0075177],"genre_scores_gemma":[0.9628262,0.0001486241,0.03524228,0.00006935214,0.00007857668,0.00008942777,0.00108152,0.00008864117,0.0003752131],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9847955,"threshold_uncertainty_score":0.08040988,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2739598167954957,"score_gpt":0.4073426254268437,"score_spread":0.133382808631348,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}