{"id":"W4409093390","doi":"10.2196/72998","title":"Citation Accuracy Challenges Posed by Large Language Models","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"scientometrics and bibliometrics research","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"Citation; Computer science; Natural language processing; Library science","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.06631449,0.001362506,0.00383508,0.004681422,0.003159006,0.01239617,0.004966582,0.004022218,0.003121694],"category_scores_gemma":[0.274838,0.001609022,0.001842866,0.007911995,0.002809564,0.01738175,0.004224836,0.006109092,0.003301159],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003760644,"about_ca_system_score_gemma":0.004137038,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01140747,"about_ca_topic_score_gemma":0.01494709,"domain_scores_codex":[0.9572923,0.02806689,0.002858749,0.004239284,0.006814923,0.000727847],"domain_scores_gemma":[0.5586054,0.4104403,0.005216117,0.01315337,0.01151198,0.001072879],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0008650725,0.0005546593,0.03513126,0.002310424,0.00249226,0.0008468851,0.00197523,0.2133314,0.002255691,0.1796464,0.09527184,0.4653189],"study_design_scores_gemma":[0.00007504065,0.00004468574,0.002664139,0.0001974256,0.0002370002,0.000309562,0.0004783192,0.6551521,0.001124739,0.3286819,0.01097548,0.00005961473],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1179938,0.02098371,0.7007673,0.118687,0.002669349,0.0003307421,0.01082117,0.007234644,0.02051224],"genre_scores_gemma":[0.8568991,0.005245142,0.1144199,0.005576847,0.003768948,0.0005112092,0.006051405,0.0009344383,0.006592963],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9953186,"threshold_uncertainty_score":0.3507087,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3294677681499928,"score_gpt":0.60707918867813,"score_spread":0.2776114205281372,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}