{"id":"W4411035583","doi":"10.1080/29974100.2025.2503343","title":"Evaluating firearm examiner testimony using large language models: a comparison of standard and knowledge-enhanced AI systems","year":2025,"lang":"en","type":"article","venue":"Journal of Psychology and AI","topic":"Digital and Cyber Forensics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; New York University Shanghai","keywords":"Computer science; Forensic engineering; Natural language processing; Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01990381,0.0008078876,0.0007704312,0.0009529577,0.0004131745,0.002566869,0.001400022,0.001359996,0.001851171],"category_scores_gemma":[0.1156586,0.0004953449,0.000717725,0.0003033202,0.0008399,0.003065106,0.002200957,0.001018929,0.0005044424],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001232791,"about_ca_system_score_gemma":0.001135677,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008321881,"about_ca_topic_score_gemma":0.001536789,"domain_scores_codex":[0.9896175,0.006430649,0.0007788613,0.001244058,0.001709109,0.0002199212],"domain_scores_gemma":[0.8837434,0.09690293,0.007247578,0.007031571,0.003802419,0.001272072],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.02204021,0.006244673,0.1179914,0.00246601,0.001774875,0.001187619,0.0131078,0.09858706,0.1030966,0.006613265,0.001936724,0.6249538],"study_design_scores_gemma":[0.002429226,0.02031017,0.1211221,0.0007329003,0.001772874,0.00194143,0.004335172,0.7293413,0.08472125,0.02438301,0.008162974,0.0007476274],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9706995,0.0003072855,0.02568952,0.0002459541,0.00003164542,0.0003694032,0.00006676949,0.00026993,0.002320027],"genre_scores_gemma":[0.9609482,0.0001257767,0.0379081,0.0002015595,0.00002481486,0.0002098573,0.00007773554,0.00002408539,0.0004797446],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01990381,"threshold_uncertainty_score":0.1052626,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06091395159346706,"score_gpt":0.4294073973453022,"score_spread":0.3684934457518351,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}