{"id":"W4391480596","doi":"10.21203/rs.3.rs-3882757/v1","title":"mCodeGPT: Enhancing Cancer Research through Zero-Shot Information Extraction from Clinical Free Text Data","year":2024,"lang":"en","type":"preprint","venue":"Research Square","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Zero (linguistics); Shot (pellet); Extraction (chemistry); Information extraction; Computer science; Cancer; Text messaging; Information retrieval; Medicine; World Wide Web; Linguistics; Materials science; Chromatography; Chemistry; Philosophy; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science","research_integrity","insufficient_payload"],"consensus_categories":["open_science"],"category_scores_codex":[0.01791444,0.0003855554,0.0005770886,0.0008495084,0.0006258594,0.002941563,0.01018767,0.0009687732,0.0002402071],"category_scores_gemma":[0.004650288,0.0003681477,0.0001940474,0.00136925,0.0002896249,0.003142241,0.04562043,0.01070274,0.001490537],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009793118,"about_ca_system_score_gemma":0.003695516,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02280623,"about_ca_topic_score_gemma":0.002631479,"domain_scores_codex":[0.987412,0.002203627,0.001546944,0.002340914,0.005073121,0.00142343],"domain_scores_gemma":[0.9851574,0.003151775,0.0002406217,0.009228716,0.001879569,0.0003418951],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000166966,0.0003846161,0.0009444677,0.004077062,0.0004741538,0.0002270283,0.01385174,0.00508595,0.001010033,0.04487725,0.2863711,0.6425296],"study_design_scores_gemma":[0.0005666019,0.0001439467,0.001290878,0.004500205,0.00002740834,0.000006614226,0.0008606394,0.5700083,0.001331709,0.3189268,0.1016198,0.0007171361],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04276994,0.008464755,0.9069004,0.01629479,0.007808891,0.002959224,0.001690238,0.0008505468,0.01226123],"genre_scores_gemma":[0.8897682,0.00781149,0.09294246,0.0002621921,0.005464748,0.0008388371,0.001490425,0.0001134258,0.001308203],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8469983,"threshold_uncertainty_score":0.999877,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4635973813089228,"score_gpt":0.5652224047004077,"score_spread":0.1016250233914849,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}