{"id":"W3021150726","doi":"10.1186/s12874-020-00991-3","title":"Feasibility and evaluation of a large-scale external validation approach for patient-level prediction in an international data network: validation of models predicting stroke in female patients newly diagnosed with atrial fibrillation","year":2020,"lang":"en","type":"article","venue":"BMC Medical Research Methodology","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"U.S. National Library of Medicine; Innovative Medicines Initiative; Health Promotion Administration, Ministry of Health and Welfare; Korea Health Industry Development Institute; European Commission; European Federation of Pharmaceutical Industries and Associations","keywords":"Observational study; Standardization; Atrial fibrillation; Computer science; Predictive modelling; Scale (ratio); Cross-validation; Medicine; Data mining; Internal medicine; Artificial intelligence; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2418694,0.001810651,0.001021137,0.001508741,0.002025842,0.003911014,0.005731431,0.002333642,0.001697116],"category_scores_gemma":[0.3260519,0.0009098476,0.002310557,0.002268655,0.00261614,0.003953935,0.0068822,0.003442335,0.0008211398],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002228349,"about_ca_system_score_gemma":0.00690207,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01220798,"about_ca_topic_score_gemma":0.009479318,"domain_scores_codex":[0.838124,0.1337789,0.006618415,0.0108418,0.00900754,0.001629315],"domain_scores_gemma":[0.6639186,0.2063181,0.01361502,0.07202958,0.0395342,0.004584508],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00658125,0.006063998,0.6607996,0.001112804,0.00466683,0.0006826701,0.006597749,0.1567034,0.007510918,0.01128128,0.01903155,0.118968],"study_design_scores_gemma":[0.003532022,0.005895277,0.2199969,0.0007736215,0.001220144,0.0006048842,0.003067737,0.719171,0.0138006,0.01462256,0.01702142,0.0002938486],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7242122,0.0003321113,0.2529579,0.00327456,0.0005198017,0.006474959,0.005004863,0.002346775,0.004876853],"genre_scores_gemma":[0.8015713,0.00007058746,0.1839729,0.000772953,0.00008982258,0.004558634,0.008166605,0.0003529784,0.0004442668],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7581306,"threshold_uncertainty_score":0.93491,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6334701342418498,"score_gpt":0.5302730048634271,"score_spread":0.1031971293784227,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}