{"id":"W6948064141","doi":"10.48448/ym5c-kv25","title":"FACTIFY3M: A benchmark for multimodal fact verification with explainability through 5W Question-Answering","year":2023,"lang":"en","type":"other","venue":"Open MIND","topic":"Species Distribution and Climate Change","field":"Environmental Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Disinformation; Benchmark (surveying); False accusation; Salient; Social media; Population; Image (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003042772,0.002828087,0.001014958,0.003731598,0.001540932,0.002640904,0.003839818,0.005395355,0.0114214],"category_scores_gemma":[0.02549358,0.0004584861,0.001818107,0.00194911,0.001147543,0.004864051,0.004205819,0.002383374,0.00681435],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002379418,"about_ca_system_score_gemma":0.001987641,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0159427,"about_ca_topic_score_gemma":0.02083129,"domain_scores_codex":[0.9945285,0.001567899,0.0004996753,0.001442103,0.00157119,0.0003905573],"domain_scores_gemma":[0.9902635,0.00538356,0.0007417663,0.002203349,0.001035825,0.0003719412],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002708834,0.001604741,0.0231571,0.006793609,0.000629545,0.002322653,0.001875855,0.04060443,0.01512762,0.02133039,0.5225674,0.3612778],"study_design_scores_gemma":[0.0007805997,0.00103115,0.02688596,0.0009438494,0.0002256465,0.002773301,0.002639911,0.517731,0.02825106,0.0398176,0.3786873,0.0002326544],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.2580491,0.0192726,0.1323215,0.007953472,0.001701528,0.00339249,0.4055564,0.1170046,0.05474832],"genre_scores_gemma":[0.2994954,0.001395058,0.2043262,0.001661775,0.0002923425,0.001564576,0.4811676,0.001507935,0.008589076],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.0159427,"threshold_uncertainty_score":0.03820831,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06054571224748662,"score_gpt":0.3368791785681661,"score_spread":0.2763334663206795,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}