{"id":"W4306405767","doi":"10.1111/jep.13779","title":"The Network Relative Model Accuracy (NeRMA) Score can quantify the relative accuracy of prediction models in concurrent external validations","year":2022,"lang":"en","type":"article","venue":"Journal of Evaluation in Clinical Practice","topic":"Mental Health Research Topics","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ottawa Hospital; University of Ottawa","funders":"","keywords":"Brier score; Predictive modelling; Computer science; Score; Random forest; Function (biology); Statistics; Calibration; Data mining; Artificial intelligence; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2559776,0.003184561,0.003382524,0.005906593,0.001507233,0.004392891,0.00267965,0.00201229,0.004623745],"category_scores_gemma":[0.541015,0.001109631,0.009331958,0.004118084,0.003672727,0.004740113,0.005224864,0.003892411,0.0006528645],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002058567,"about_ca_system_score_gemma":0.002696166,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003045626,"about_ca_topic_score_gemma":0.004555305,"domain_scores_codex":[0.7912629,0.1666991,0.01424604,0.01539477,0.0114306,0.000966684],"domain_scores_gemma":[0.2950879,0.6316715,0.02356278,0.03838255,0.01005612,0.001239094],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.006560532,0.0004655358,0.4839365,0.004767407,0.05777124,0.0006689723,0.001905806,0.261925,0.001869905,0.02517165,0.01017076,0.1447867],"study_design_scores_gemma":[0.001379412,0.003907004,0.1077211,0.002499851,0.01844568,0.001083286,0.0005742089,0.7330429,0.006146423,0.1091934,0.01540637,0.0006003345],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2865165,0.008086962,0.6793634,0.003039051,0.0008869884,0.002034567,0.005235189,0.002208696,0.0126286],"genre_scores_gemma":[0.9052328,0.0004702248,0.08985039,0.0004503576,0.000121603,0.001358309,0.001741167,0.0003477357,0.0004274402],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.7440224,"threshold_uncertainty_score":0.9175121,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4905445628808434,"score_gpt":0.615057805187723,"score_spread":0.1245132423068796,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}