{"id":"W4200359287","doi":"10.2196/preprints.35621","title":"Rule-based Natural Language Processing for Automation of Stroke Data Extraction: A Validation Study (Preprint)","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Acute Ischemic Stroke Management","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Medicine; Occlusion; Collateral circulation; Stroke (engine); Angiography; Artificial intelligence; Radiology; Natural language processing; Computer science; Internal medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01948072,0.0005466067,0.0004592729,0.001258067,0.0004045307,0.00123114,0.0009758278,0.0005639102,0.001994922],"category_scores_gemma":[0.05134768,0.000280389,0.00111873,0.0008586506,0.0004986487,0.0009821745,0.0007690808,0.0004503989,0.001250252],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006001524,"about_ca_system_score_gemma":0.001307798,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004152786,"about_ca_topic_score_gemma":0.002410113,"domain_scores_codex":[0.9915493,0.005137137,0.00113878,0.0008618791,0.001119472,0.0001934251],"domain_scores_gemma":[0.9422995,0.04192767,0.001720183,0.004374583,0.009287553,0.0003904903],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004378385,0.007982335,0.2828674,0.002760095,0.001066685,0.001150479,0.004722851,0.01685538,0.02714572,0.0007464408,0.008886389,0.6414379],"study_design_scores_gemma":[0.002229106,0.02168288,0.609068,0.001041139,0.001741153,0.003043368,0.003873465,0.2242159,0.1068521,0.001601642,0.02432732,0.0003240238],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.9398571,0.0008988354,0.05032515,0.0002572501,0.000130333,0.002618092,0.002576117,0.0009950415,0.002342195],"genre_scores_gemma":[0.8499445,0.0007187375,0.1383952,0.0002650452,0.0001064297,0.001991851,0.006853413,0.0001704127,0.001554507],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01948072,"threshold_uncertainty_score":0.1030251,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04424155444754947,"score_gpt":0.3662642032599829,"score_spread":0.3220226488124335,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}