{"id":"W4378213586","doi":"10.2196/preprints.45662","title":"Generate Analysis-Ready Data for Real-world Evidence: Tutorial for Harnessing Electronic Health Records With Advanced Informatic Technologies (Preprint)","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; University of British Columbia","funders":"","keywords":"Computer science; Leverage (statistics); Unstructured data; Data science; Big data; Data mining; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01189756,0.002868202,0.00161861,0.004934259,0.0007074148,0.005672667,0.003094416,0.002141216,0.06873068],"category_scores_gemma":[0.03997723,0.002402717,0.003556446,0.003566514,0.0009768844,0.005447048,0.004212332,0.005164996,0.05220586],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001295001,"about_ca_system_score_gemma":0.002863738,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0022357,"about_ca_topic_score_gemma":0.002676341,"domain_scores_codex":[0.9960238,0.001586692,0.0006803746,0.0004218678,0.00120203,0.00008519514],"domain_scores_gemma":[0.9728546,0.0212317,0.0009230617,0.001816347,0.002523167,0.0006512552],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001690121,0.0001446139,0.001096358,0.005473845,0.0003341475,0.0008153574,0.0009653253,0.008773027,0.006827106,0.05049573,0.2968361,0.6280693],"study_design_scores_gemma":[0.0001942249,0.0001088944,0.001658812,0.002685191,0.0001077501,0.001267712,0.0003143439,0.04754087,0.008554945,0.1309978,0.8063703,0.0001992018],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0002943114,0.001609832,0.964799,0.001695412,0.0004293676,0.001109379,0.005590303,0.02156337,0.00290901],"genre_scores_gemma":[0.001601355,0.001779014,0.9852963,0.0006493684,0.0002293018,0.0008496189,0.003725975,0.002975109,0.00289391],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.06873068,"threshold_uncertainty_score":0.2299271,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1230525538215113,"score_gpt":0.4146534769718794,"score_spread":0.2916009231503682,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}