{"id":"W4412889118","doi":"10.18653/v1/2025.bionlp-1.12","title":"Error Detection in Medical Note through Multi Agent Debate","year":2025,"lang":"en","type":"article","venue":"","topic":"Pharmacy and Medical Practices","field":"Pharmacology, Toxicology and Pharmaceutics","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"UK Research and Innovation","keywords":"Computer science; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001442306,0.0001720563,0.0002243428,0.0001163751,0.0001666366,0.00001155384,0.0002851564,0.0005186004,0.007086861],"category_scores_gemma":[0.001075726,0.000146276,0.0000673246,0.0003675929,0.0001962244,0.0002596699,0.0001198899,0.00176358,0.0004007183],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001020139,"about_ca_system_score_gemma":0.000204117,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002652266,"about_ca_topic_score_gemma":0.0009451904,"domain_scores_codex":[0.9979157,0.000660091,0.0004276204,0.0003191475,0.0002395397,0.0004378497],"domain_scores_gemma":[0.9985673,0.0009481632,0.00007541771,0.0001340164,0.00003494533,0.0002401333],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001836039,0.003310282,0.00572534,0.0002672281,0.0003927741,0.001123447,0.001878903,0.0003083596,0.0416671,0.004376835,0.02968527,0.9094284],"study_design_scores_gemma":[0.003873709,0.00003586576,0.00146155,0.00002819288,0.00008692786,0.00002281842,0.0001146949,0.04391051,0.08339331,0.0006362957,0.8662692,0.0001668895],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5677304,0.003137296,0.07164515,0.06309601,0.01713792,0.001647731,0.00002373978,0.0007134585,0.2748682],"genre_scores_gemma":[0.9510081,0.0009358965,0.0003807734,0.04309486,0.000136107,0.00006664837,0.000006869157,0.000009375704,0.004361381],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9092615,"threshold_uncertainty_score":0.9938208,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2128258572798672,"score_gpt":0.5529093905737874,"score_spread":0.3400835332939203,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}