{"id":"W4412889118","doi":"10.18653/v1/2025.bionlp-1.12","title":"Error Detection in Medical Note through Multi Agent Debate","year":2025,"lang":"en","type":"article","venue":"","topic":"Pharmacy and Medical Practices","field":"Pharmacology, Toxicology and Pharmaceutics","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"UK Research and Innovation","keywords":"Computer science; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01706361,0.0009389239,0.001302948,0.002067701,0.001724104,0.00450661,0.002612989,0.002873579,0.002781251],"category_scores_gemma":[0.08043582,0.0006279076,0.0009648689,0.0008300968,0.001957304,0.005596306,0.005582696,0.00272776,0.0009790559],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001957311,"about_ca_system_score_gemma":0.003193967,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003017548,"about_ca_topic_score_gemma":0.003067882,"domain_scores_codex":[0.9813261,0.01121227,0.001330687,0.002300583,0.003244271,0.000586131],"domain_scores_gemma":[0.921047,0.05604143,0.009011604,0.006752624,0.005984965,0.001162377],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002560448,0.0009801936,0.04502433,0.0009154702,0.0003879439,0.001968368,0.01144085,0.1949378,0.02278865,0.0509329,0.008279455,0.6597837],"study_design_scores_gemma":[0.0001082351,0.0002111986,0.001861282,0.00009886077,0.00007773218,0.0002837292,0.0007651392,0.9497888,0.01358392,0.02647625,0.006666424,0.0000783248],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1890254,0.0006753416,0.7940502,0.003260399,0.0001358028,0.000524269,0.0002147811,0.005974452,0.006139352],"genre_scores_gemma":[0.7628276,0.0001069297,0.2339402,0.0004609124,0.0000717822,0.0001361513,0.0002225637,0.0001876105,0.002046255],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01706361,"threshold_uncertainty_score":0.09024203,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2128258572798672,"score_gpt":0.5529093905737874,"score_spread":0.3400835332939203,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}