{"id":"W4390189903","doi":"10.1109/iccvw60793.2023.00465","title":"Confusing Large Models by Confusing Small Models","year":2023,"lang":"en","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Robustness (evolution); Machine learning; Heuristics; Confusion; Artificial intelligence; Benchmark (surveying); Focus (optics); Upsampling; Image (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008120107,0.002359314,0.001724878,0.001696285,0.001080953,0.00279819,0.001767029,0.002216164,0.001987341],"category_scores_gemma":[0.03717411,0.0008096861,0.001386291,0.000819559,0.003698749,0.004464421,0.00482143,0.00519152,0.0007300705],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001502143,"about_ca_system_score_gemma":0.0007727101,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0026217,"about_ca_topic_score_gemma":0.003453443,"domain_scores_codex":[0.994662,0.002378602,0.000249248,0.001188572,0.001144947,0.0003766756],"domain_scores_gemma":[0.9788529,0.01213398,0.001535169,0.00552506,0.001248822,0.0007040676],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008323963,0.000178581,0.01370864,0.000259982,0.0004965013,0.0005254106,0.0007168609,0.8013951,0.01763567,0.02353661,0.00634304,0.1343711],"study_design_scores_gemma":[0.00001714145,0.0002031123,0.002338419,0.00005917945,0.00006985429,0.0003711354,0.0001246567,0.9405708,0.01200902,0.04209142,0.002084789,0.00006047605],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3069751,0.001615228,0.6783299,0.002790319,0.0003424531,0.0001757367,0.0003615098,0.002444704,0.006965005],"genre_scores_gemma":[0.9038601,0.0003248601,0.0912527,0.001018227,0.000207971,0.0000870585,0.0006267283,0.0005047657,0.002117587],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008120107,"threshold_uncertainty_score":0.04294372,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04520528662869609,"score_gpt":0.2644070661826076,"score_spread":0.2192017795539115,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}