{"id":"W4323651448","doi":"10.48550/arxiv.2303.04075","title":"Exploiting Trust for Resilient Hypothesis Testing with Malicious Robots (evolved version)","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Air Force Office of Scientific Research; Canadian Institute for Advanced Research","keywords":"Robot; Exploit; Computer science; Adversarial system; Artificial intelligence; Communication source; Task (project management); Statistical hypothesis testing; Computer security; Machine learning; Computer network; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009067711,0.001253521,0.001551444,0.001097455,0.0006265384,0.001647396,0.003197186,0.002111258,0.001833992],"category_scores_gemma":[0.04007654,0.0006990404,0.001586199,0.0007227052,0.003402302,0.003141913,0.005148242,0.003017833,0.0005787621],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001715142,"about_ca_system_score_gemma":0.001807113,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001927239,"about_ca_topic_score_gemma":0.00102112,"domain_scores_codex":[0.9932812,0.00274257,0.000319174,0.001233935,0.001838135,0.0005848765],"domain_scores_gemma":[0.9807827,0.01241204,0.001829642,0.002911826,0.001566817,0.0004969319],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004147645,0.0001076081,0.002329303,0.0001436246,0.0001708246,0.0004665816,0.0003361839,0.8122551,0.005563986,0.09405193,0.001334197,0.08282595],"study_design_scores_gemma":[0.00002369877,0.0000623321,0.0001359228,0.00001047817,0.00001171526,0.00005952804,0.000009587448,0.9673752,0.001530454,0.03038122,0.0003869258,0.00001299535],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008442741,0.0001129961,0.9900674,0.0002494846,0.00002711007,0.00005179293,0.00003821165,0.0003661361,0.0006442075],"genre_scores_gemma":[0.6732546,0.0001570093,0.3239022,0.0003947097,0.0001137207,0.0002800995,0.0001619312,0.0001206097,0.001615082],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009067711,"threshold_uncertainty_score":0.04795521,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1575077665701807,"score_gpt":0.2114097435912075,"score_spread":0.05390197702102678,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}