{"id":"W4408666991","doi":"10.70777/si.v2i1.13799","title":"Can a Bayesian Oracle Prevent Harm from an Agent?","year":2025,"lang":"en","type":"article","venue":"SuperIntelligence - Robotics - Safety & Alignment","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Harm; Oracle; Bayesian probability; Computer science; Psychology; Artificial intelligence; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0184691,0.001137955,0.002159758,0.00128055,0.001545293,0.004206204,0.003177151,0.005599656,0.0136443],"category_scores_gemma":[0.09557106,0.001049053,0.001165764,0.0007929425,0.007404822,0.01624293,0.003712708,0.006838924,0.002217385],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002054429,"about_ca_system_score_gemma":0.003448861,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002453704,"about_ca_topic_score_gemma":0.002993253,"domain_scores_codex":[0.9913782,0.004180452,0.0003553347,0.001365964,0.001833722,0.0008862792],"domain_scores_gemma":[0.9454551,0.04173554,0.002937689,0.006133657,0.002285543,0.001452445],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004253058,0.0001458364,0.00320026,0.0002505932,0.0001168819,0.0002367501,0.0003818005,0.0710066,0.001023633,0.8373515,0.008669497,0.07719139],"study_design_scores_gemma":[0.00006835168,0.00006957946,0.0003816632,0.0000719381,0.00002666204,0.0001658231,0.00008609559,0.1412913,0.0006892527,0.8525055,0.00460882,0.00003501626],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01909738,0.0008476827,0.9291012,0.02527256,0.0002656886,0.0001355561,0.0003071352,0.0007629013,0.02420998],"genre_scores_gemma":[0.8305057,0.0009149179,0.1551832,0.003921055,0.0005606281,0.0002551656,0.0002498403,0.0002178764,0.008191587],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0184691,"threshold_uncertainty_score":0.09767509,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01427476112546157,"score_gpt":0.2769327674598674,"score_spread":0.2626580063344058,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}