{"id":"W4387357939","doi":"10.2196/44892","title":"A Multilabel Text Classifier of Cancer Literature at the Publication Level: Methods Study of Medical Text Classification","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Classifier (UML); Artificial intelligence; Machine learning; Terminology; Information retrieval; Natural language processing","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002067364,0.0001652563,0.0002908435,0.0001163911,0.00009533508,0.00002398496,0.0007422709,0.0007024293,0.0002542767],"category_scores_gemma":[0.00227577,0.00009950679,0.00009387221,0.0007469894,0.0005464604,0.00001067438,0.0004824763,0.0003777314,0.00002129011],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002363061,"about_ca_system_score_gemma":0.0002908649,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003374491,"about_ca_topic_score_gemma":0.0001292018,"domain_scores_codex":[0.9970739,0.0002310182,0.0009229885,0.0001804508,0.001326917,0.0002646708],"domain_scores_gemma":[0.9982598,0.0002374162,0.0004210506,0.0005644049,0.0003062644,0.0002111057],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000106286,0.0005720464,0.003893049,0.0003535639,0.0002406569,0.00000250135,0.01190902,0.000003775547,0.006167851,0.000254181,0.1225361,0.853961],"study_design_scores_gemma":[0.006632533,0.001824963,0.1321019,0.0008267629,0.00018136,0.00005378509,0.04715971,0.06417367,0.02104046,0.00013197,0.7250636,0.0008093384],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9873823,0.0007744469,0.002542926,0.007125228,0.0003561406,0.0006276293,0.00007592911,0.00006450582,0.001050923],"genre_scores_gemma":[0.9914129,0.001264172,0.002678128,0.001197281,0.0002509756,0.0004197475,0.0004589788,0.00002627183,0.002291563],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8531516,"threshold_uncertainty_score":0.5417778,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08301918992268613,"score_gpt":0.4361605030102835,"score_spread":0.3531413130875974,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}