{"id":"W4409716856","doi":"10.1111/jels.12413","title":"Hallucination‐Free? Assessing the Reliability of Leading <scp>AI</scp> Legal Research Tools","year":2025,"lang":"en","type":"article","venue":"Journal of Empirical Legal Studies","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":92,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Stanford Law School; University of Toronto; Fordham University","keywords":"Hallucinating; Automatic summarization; Legal psychology; Typology; Lexis; Computer science; Legal case; Legal research; Psychology; Data science; Artificial intelligence; Social psychology; Law; Political science; Sociology; Linguistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.08605786,0.0006704377,0.0009030268,0.006744006,0.001584683,0.007902852,0.00309882,0.002631462,0.002425232],"category_scores_gemma":[0.53018,0.0006317749,0.0006143615,0.003596926,0.003712983,0.007832309,0.007094907,0.002074446,0.001663334],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002147727,"about_ca_system_score_gemma":0.001948739,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004348563,"about_ca_topic_score_gemma":0.003182405,"domain_scores_codex":[0.8916167,0.05578578,0.01127214,0.008499637,0.0307736,0.002052121],"domain_scores_gemma":[0.2797127,0.5692166,0.06136611,0.04578853,0.03852417,0.005391843],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003379101,0.001652304,0.4795417,0.003120526,0.001064732,0.0007612717,0.04870131,0.01646373,0.009944828,0.005643711,0.01160913,0.4181177],"study_design_scores_gemma":[0.0005695639,0.007892582,0.6049523,0.002070366,0.001337861,0.002535345,0.03933703,0.2362376,0.04694758,0.02011336,0.03703967,0.0009666422],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9670214,0.0008045329,0.01378294,0.001593029,0.00009264274,0.0005635216,0.0006428939,0.002123374,0.01337563],"genre_scores_gemma":[0.9902462,0.0001295738,0.007894843,0.0002095241,0.00004316503,0.0001276852,0.0005404877,0.00012056,0.0006878562],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9139422,"threshold_uncertainty_score":0.4551229,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3252983431727126,"score_gpt":0.5757539876260013,"score_spread":0.2504556444532887,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}