{"id":"W4407697469","doi":"10.1016/j.inffus.2025.104092","title":"VLDBench Evaluating multimodal disinformation with regulatory alignment","year":2025,"lang":"en","type":"preprint","venue":"Information Fusion","topic":"Misinformation and Its Impacts","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University; Toronto Zoo; Vector Institute","funders":"","keywords":"Disinformation; Benchmark (surveying); Computer science; Artificial intelligence; Natural language processing; Geography; World Wide Web; Cartography; Social media","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01298764,0.001543006,0.001603174,0.004511525,0.0009783277,0.00483938,0.001465783,0.002812283,0.009557025],"category_scores_gemma":[0.0544591,0.0004466301,0.000788784,0.003603692,0.001374364,0.004661409,0.00348233,0.00158971,0.001787524],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002491646,"about_ca_system_score_gemma":0.003851249,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01750505,"about_ca_topic_score_gemma":0.01000305,"domain_scores_codex":[0.9908853,0.004906458,0.0003625605,0.001162889,0.002085945,0.0005968668],"domain_scores_gemma":[0.9736435,0.01937393,0.0008668138,0.002232192,0.003297198,0.0005863393],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003869004,0.0009987679,0.01806665,0.0007926176,0.000637041,0.0002379954,0.0003560628,0.490782,0.007753211,0.02264691,0.01523738,0.4386223],"study_design_scores_gemma":[0.00007527701,0.0004344685,0.002777542,0.00006000136,0.0001305864,0.00005239039,0.0002939055,0.9616644,0.01093582,0.02062427,0.002905803,0.00004550289],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.406403,0.004534732,0.5312603,0.003765439,0.0007302114,0.0005574797,0.004811891,0.007812154,0.04012477],"genre_scores_gemma":[0.9183198,0.0003296721,0.07267129,0.0003182501,0.0001007301,0.0001369378,0.003063054,0.0002891748,0.004771182],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01750505,"threshold_uncertainty_score":0.06868607,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02982611508341688,"score_gpt":0.3501096435763882,"score_spread":0.3202835284929713,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}