{"id":"W6906393256","doi":"10.17605/osf.io/4k69q","title":"Comparing the National Library of Medicine (NLM)’s Medical Text Indexer (MTI) to Human Indexing: A Pilot Study","year":2022,"lang":"en","type":"article","venue":"OSF Preprints (OSF Preprints)","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"National library; Medical library; MEDLINE; Medical journal; Digital library","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["metaresearch"],"domain":"methods","study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.00464698,0.0002069747,0.0003426619,0.0001281564,0.0003583296,0.00002373039,0.0017055,0.0001030576,0.1083328],"category_scores_gemma":[0.003074768,0.0001686002,0.00007765705,0.0002927375,0.0004661233,0.000007787847,0.00456494,0.0005311161,0.004990684],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004766885,"about_ca_system_score_gemma":0.0002445185,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000168337,"about_ca_topic_score_gemma":0.00006262276,"domain_scores_codex":[0.9961625,0.0007681217,0.0005827534,0.001097537,0.001092921,0.0002961079],"domain_scores_gemma":[0.9980568,0.0002340012,0.0001979466,0.001219876,0.00008255414,0.0002087971],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00141109,0.00458963,0.5249692,0.00009221363,0.0008486935,0.00004676082,0.00573297,0.001499966,0.09424356,0.002580316,0.3473445,0.01664115],"study_design_scores_gemma":[0.005162463,0.0004636649,0.3725361,0.0001201542,0.0001085457,0.0001435029,0.004870957,0.0004458648,0.01401302,0.003363986,0.5979712,0.0008004931],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8383788,0.000007711329,0.0003110854,0.002889668,0.0001978597,0.000590454,0.00001016683,0.00005184134,0.1575624],"genre_scores_gemma":[0.9736765,0.000008417665,0.0001657209,0.001203989,0.0001994471,0.0002606586,0.0000388311,0.00002606309,0.02442039],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2506267,"threshold_uncertainty_score":0.995784,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04820279487776538,"score_gpt":0.3148710831137579,"score_spread":0.2666682882359925,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}