{"id":"W4407193212","doi":"10.3390/ai6020029","title":"Priv-IQ: A Benchmark and Comparative Evaluation of Large Multimodal Models on Privacy Competencies","year":2025,"lang":"en","type":"article","venue":"AI","topic":"Technology Adoption and User Behaviour","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"Benchmark (surveying); Computer science; Artificial intelligence; Geography; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01198348,0.002178398,0.0008908584,0.002691504,0.0007392619,0.002005728,0.002355083,0.002038001,0.004617232],"category_scores_gemma":[0.03732738,0.0003594206,0.001150497,0.001577252,0.001108011,0.003887877,0.004621913,0.001916324,0.001482789],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00211795,"about_ca_system_score_gemma":0.001886809,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01176598,"about_ca_topic_score_gemma":0.0122993,"domain_scores_codex":[0.9895135,0.006360364,0.0008021841,0.001131951,0.001785499,0.0004063926],"domain_scores_gemma":[0.977336,0.01506843,0.0009039215,0.003164001,0.002558843,0.0009687758],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003097283,0.005039411,0.05881763,0.002493174,0.001275248,0.0004257381,0.002094218,0.4550207,0.009758895,0.008920982,0.03998228,0.4130744],"study_design_scores_gemma":[0.0003474918,0.002717552,0.02497951,0.0002465847,0.0001818243,0.0002348158,0.001033321,0.9404449,0.01176474,0.006394997,0.01152143,0.0001328409],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8753768,0.002302582,0.07191829,0.001224487,0.0002799902,0.001233916,0.009818013,0.01168555,0.02616035],"genre_scores_gemma":[0.9032474,0.0004772453,0.06801809,0.0003447966,0.00007255664,0.0007789241,0.02316654,0.0006878814,0.003206449],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01198348,"threshold_uncertainty_score":0.06337547,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2389025128290721,"score_gpt":0.4638331165797938,"score_spread":0.2249306037507217,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}