{"id":"W2417098700","doi":"10.1016/j.jclinepi.2016.05.007","title":"Geographic and temporal validity of prediction models: different approaches were useful to examine model performance","year":2016,"lang":"en","type":"article","venue":"Journal of Clinical Epidemiology","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":85,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Sunnybrook Health Science Centre; Institute for Clinical Evaluative Sciences","funders":"National Institute of Neurological Disorders and Stroke; Canadian Institutes of Health Research; Nederlandse Organisatie voor Wetenschappelijk Onderzoek","keywords":"Computer science; Predictive modelling; Statistics; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2946886,0.003296349,0.003039927,0.004629931,0.00171756,0.006918091,0.003681121,0.002914365,0.002292176],"category_scores_gemma":[0.4657099,0.001373537,0.01320891,0.007174911,0.003789707,0.0084449,0.006991261,0.005035308,0.0003167629],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002577119,"about_ca_system_score_gemma":0.003988893,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005945751,"about_ca_topic_score_gemma":0.005252355,"domain_scores_codex":[0.7664238,0.1833719,0.02020107,0.01814706,0.01019657,0.001659603],"domain_scores_gemma":[0.4790953,0.4212381,0.0345307,0.04946346,0.01474521,0.0009273111],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003841218,0.0003957628,0.6797056,0.003182346,0.06624643,0.0006559116,0.003093322,0.1160694,0.001428206,0.01369371,0.002745378,0.1089428],"study_design_scores_gemma":[0.001115108,0.005078462,0.3840595,0.003150002,0.02751674,0.001561629,0.003112891,0.4594198,0.008185213,0.08947141,0.01651574,0.0008134784],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3087803,0.01833708,0.6558605,0.00440226,0.0008710903,0.001567426,0.003910813,0.0007753322,0.005495185],"genre_scores_gemma":[0.9437955,0.001280666,0.05048047,0.0005144062,0.0002345915,0.001075393,0.001929923,0.0002523469,0.000436724],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7053114,"threshold_uncertainty_score":0.8697746,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4903653697627472,"score_gpt":0.4388877148245395,"score_spread":0.05147765493820777,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}