{"id":"W4224214111","doi":"10.21203/rs.3.rs-1521400/v1","title":"The Network Relative Model Accuracy (Nerma) Score Can Quantify the Relative Accuracy of Prediction Models in Concurrent External Validations","year":2022,"lang":"en","type":"preprint","venue":"Research Square","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ottawa Hospital","funders":"","keywords":"Brier score; Computer science; Predictive modelling; Score; Function (biology); Random forest; Calibration; Statistics; Data mining; Artificial intelligence; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2967891,0.00298714,0.003613272,0.007667148,0.001387965,0.005073984,0.002654192,0.002040558,0.00458619],"category_scores_gemma":[0.6147636,0.001184546,0.01199628,0.005459116,0.003104259,0.005487804,0.005141911,0.00359996,0.0006521838],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001982805,"about_ca_system_score_gemma":0.002697831,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002629946,"about_ca_topic_score_gemma":0.003949353,"domain_scores_codex":[0.7083284,0.2332636,0.02373734,0.0184287,0.01514849,0.001093536],"domain_scores_gemma":[0.2338023,0.6867385,0.02738258,0.04006476,0.01105906,0.000952865],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.007237521,0.0003997459,0.425319,0.008748377,0.09419291,0.0008042226,0.001775214,0.2530504,0.00216141,0.02965349,0.01071917,0.1659385],"study_design_scores_gemma":[0.001405945,0.003457698,0.08761612,0.003628639,0.03157925,0.001104486,0.0005497088,0.7302618,0.006864196,0.1155129,0.0173722,0.0006470731],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2212301,0.01148682,0.7397473,0.003004733,0.001029777,0.002352323,0.006177683,0.002400048,0.01257123],"genre_scores_gemma":[0.8846719,0.0005878279,0.1100044,0.0004506921,0.0001407894,0.001548521,0.001830627,0.0003256309,0.0004397793],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.7032108,"threshold_uncertainty_score":0.8671842,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8402462001581907,"score_gpt":0.608716174131749,"score_spread":0.2315300260264417,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}