{"id":"W4408181626","doi":"10.22541/essoar.174129298.81850235/v1","title":"Defining an independent reference model for event detection skill scores","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Risk and Safety Analysis","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Nuclear Safety and Security Commission; Alberta Machine Intelligence Institute; National Aeronautics and Space Administration; University of Alberta; National Science Foundation","keywords":"Event (particle physics); Computer science; Psychology; Econometrics; Cognitive psychology; Mathematics; Physics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0245188,0.00234713,0.002134183,0.004442861,0.001070373,0.005469207,0.009095995,0.002860689,0.008070149],"category_scores_gemma":[0.08300761,0.00104655,0.003485443,0.002261466,0.003698983,0.008102248,0.005632353,0.006073036,0.003514488],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004267758,"about_ca_system_score_gemma":0.003683827,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006416752,"about_ca_topic_score_gemma":0.004946268,"domain_scores_codex":[0.9817355,0.006346526,0.001028891,0.004048344,0.005858205,0.0009825666],"domain_scores_gemma":[0.9552984,0.01883623,0.003385233,0.01017484,0.01136165,0.0009437173],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002804851,0.0003977361,0.0102891,0.0002072885,0.0003130847,0.0002882811,0.0006952558,0.3639828,0.003667664,0.5296876,0.007059093,0.08313169],"study_design_scores_gemma":[0.00005089379,0.0002473998,0.002576725,0.00009106833,0.00007643853,0.0001220448,0.0001279963,0.800301,0.003234669,0.1841156,0.008920703,0.0001356517],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.007869081,0.00004481506,0.9858398,0.0002227547,0.00005368225,0.0001638091,0.0002469826,0.0007818915,0.004777062],"genre_scores_gemma":[0.3051906,0.0001660181,0.6821706,0.0003374409,0.0001380066,0.001813795,0.002168774,0.0009919592,0.007022891],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0245188,"threshold_uncertainty_score":0.1296694,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1426862218944084,"score_gpt":0.4315118414733023,"score_spread":0.288825619578894,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}