{"id":"W6977275162","doi":"10.6084/m9.figshare.16909627","title":"Additional file 1 of Assessing the suitability of general practice electronic health records for clinical prediction model development: a data quality assessment","year":2021,"lang":"en","type":"article","venue":"Figshare","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Data quality; Coding (social sciences); Health records; Health data; General practice; Quality (philosophy); Data collection; Electronic health record","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002066523,0.0001090745,0.0002722971,0.00002611656,0.0002316125,0.00006600856,0.0009508089,0.00008263507,0.2220172],"category_scores_gemma":[0.03260337,0.00009513873,0.00007823906,0.0002663942,0.00001365249,0.0006757735,0.0007448876,0.0004660579,0.00000801029],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002018855,"about_ca_system_score_gemma":0.01075075,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003043394,"about_ca_topic_score_gemma":0.00009143184,"domain_scores_codex":[0.9964823,0.001179274,0.0009524937,0.000579559,0.0004996387,0.0003067092],"domain_scores_gemma":[0.9843828,0.01209974,0.001081377,0.001355956,0.0009930272,0.00008709369],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004533319,0.0001436995,0.0001304377,0.0003472583,0.00002472664,2.735667e-7,0.00008133698,0.0003505816,4.694803e-7,0.0002905112,0.9669982,0.03162801],"study_design_scores_gemma":[0.0001156467,0.00006832441,0.0387874,0.0004608623,0.000002801938,0.000005971491,0.0000328891,0.507172,0.000004799076,0.0002361622,0.4530416,0.00007146414],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.00003644229,0.00009954545,0.02025012,0.001763983,0.00005981582,0.0003402006,0.9770503,0.0000459638,0.0003536425],"genre_scores_gemma":[0.001420803,0.000003733128,0.3090076,0.0004283918,0.00009181954,0.000375454,0.6885709,0.000007845369,0.00009345796],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.5139565,"threshold_uncertainty_score":0.9948574,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3020492965549068,"score_gpt":0.5126365692006355,"score_spread":0.2105872726457287,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}