{"id":"W4392656250","doi":"10.1186/s12874-024-02192-8","title":"Text analysis framework for identifying mutations among non-small cell lung cancer patients from laboratory data","year":2024,"lang":"en","type":"article","venue":"BMC Medical Research Methodology","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Calgary Laboratory Services; University of Calgary","funders":"Canadian Institutes of Health Research","keywords":"Computer science; Lexical analysis; Natural language processing; Artificial intelligence; Information extraction; Syntax; Context (archaeology); Population; Information retrieval; Medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.008110223,0.0001768657,0.0004189103,0.00032306,0.0001976553,0.00008525437,0.001211491,0.0008554631,0.0004376504],"category_scores_gemma":[0.04493637,0.0001462771,0.0001658475,0.0009987368,0.0007718177,0.00000843697,0.0009207528,0.0007664572,0.00001516232],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004180392,"about_ca_system_score_gemma":0.001398428,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008112251,"about_ca_topic_score_gemma":0.002662915,"domain_scores_codex":[0.9944894,0.002602213,0.0004322951,0.001068941,0.0007377514,0.0006693996],"domain_scores_gemma":[0.9840954,0.01419212,0.0000705738,0.0008666943,0.0003475028,0.0004277422],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0008063516,0.0009111928,0.5090109,0.00306979,0.006898306,0.0001458228,0.002853213,0.000131485,0.03886734,0.001043512,0.1404155,0.2958466],"study_design_scores_gemma":[0.006230558,0.0021881,0.3590809,0.002544032,0.005521245,0.000004381719,0.00715178,0.271979,0.07596464,0.02164987,0.2447876,0.00289786],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3403436,0.0069747,0.6508814,0.0002932874,0.0005842232,0.0002476471,0.0006030305,0.00002891942,0.00004320733],"genre_scores_gemma":[0.4378095,0.001710348,0.5536845,0.0003489322,0.001593618,0.0005343574,0.003670315,0.00006403657,0.0005844233],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2929488,"threshold_uncertainty_score":0.9631085,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.354800330344782,"score_gpt":0.5470634171707748,"score_spread":0.1922630868259927,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}