{"id":"W4406352685","doi":"10.5195/jmla.2025.1936","title":"Algorithmic indexing in MEDLINE frequently overlooks important concepts and may compromise literature search results","year":2025,"lang":"en","type":"article","venue":"Journal of the Medical Library Association JMLA","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Public Health Agency of Canada; Université de Montréal","funders":"McGill University Health Centre; McGill University","keywords":"Search engine indexing; Information retrieval; MEDLINE; Computer science; Medical record; Subject (documents); Index (typography); Data mining; Medicine; Library science; World Wide Web; Radiology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","scholarly_communication"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2712946,0.001480688,0.00336716,0.03254696,0.003423712,0.01535205,0.004237161,0.00211649,0.003936142],"category_scores_gemma":[0.6683505,0.001301354,0.002638713,0.03237342,0.005294185,0.0120943,0.007333483,0.001776133,0.002584913],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006077366,"about_ca_system_score_gemma":0.01694313,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00329302,"about_ca_topic_score_gemma":0.005408872,"domain_scores_codex":[0.6305242,0.1979645,0.1025863,0.007308002,0.05996378,0.001653352],"domain_scores_gemma":[0.2798276,0.5613331,0.07424514,0.03345203,0.04978503,0.001357164],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001776926,0.0003588598,0.06064709,0.03788686,0.001627406,0.0008833209,0.01403027,0.004102997,0.007607393,0.03577589,0.02547192,0.8098311],"study_design_scores_gemma":[0.002001764,0.003851081,0.1729143,0.08826359,0.005190854,0.01116245,0.02605829,0.03672466,0.02262796,0.2161869,0.4133962,0.001621973],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.266484,0.1111353,0.4481488,0.05844226,0.004482872,0.02631512,0.007979586,0.006766876,0.07024524],"genre_scores_gemma":[0.3353962,0.01994488,0.6176063,0.008974426,0.00152626,0.009312812,0.004140909,0.0008279998,0.002270221],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9846479,"threshold_uncertainty_score":0.8986235,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.005347617935268055,"score_gpt":0.2804653613786845,"score_spread":0.2751177434434164,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}