{"id":"W4409212246","doi":"10.1111/1468-5973.70044","title":"Evaluating the Performance of Agreement Metrics in a Delphi Study on Chemical, Biological, Radiological and Nuclear Major Incidents Preparedness Using Classical and Machine Learning Approaches","year":2025,"lang":"en","type":"article","venue":"Journal of Contingencies and Crisis Management","topic":"Risk and Safety Analysis","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Radiological weapon; Preparedness; Delphi; Delphi method; Engineering; Computer science; Artificial intelligence; Chemistry; Political science; Radiochemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2813133,0.00110692,0.00120184,0.007008764,0.002977702,0.003221674,0.001568617,0.001172069,0.001887978],"category_scores_gemma":[0.3559161,0.0008290641,0.001719674,0.003711723,0.002862727,0.004428145,0.007739104,0.001401864,0.00040104],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005025152,"about_ca_system_score_gemma":0.005230126,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002021512,"about_ca_topic_score_gemma":0.002252681,"domain_scores_codex":[0.7411652,0.2142525,0.01806545,0.005002535,0.01737413,0.004140272],"domain_scores_gemma":[0.4286843,0.4852478,0.0161326,0.008394117,0.0583987,0.003142534],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007065577,0.002188134,0.2097757,0.004081761,0.001070622,0.0004495329,0.4140374,0.0150482,0.01306487,0.01171586,0.003391097,0.3181112],"study_design_scores_gemma":[0.001040798,0.02014634,0.344723,0.002712544,0.0006388981,0.0006802241,0.3995838,0.1651242,0.0302681,0.02150603,0.01241447,0.001161686],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9641722,0.0001289565,0.02807441,0.0003418213,0.00005064401,0.002362514,0.0001466247,0.00008819871,0.004634567],"genre_scores_gemma":[0.9675085,0.00006975191,0.0286305,0.00007094926,0.00001011039,0.003270687,0.00009674732,0.0000268634,0.0003159021],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7186867,"threshold_uncertainty_score":0.8862686,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2746546593058555,"score_gpt":0.4066631976442115,"score_spread":0.132008538338356,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}