{"id":"W6910477341","doi":"10.48448/2zwc-6969","title":"A New Benchmark and Model for Challenging Image Manipulation Detection","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Benchmark (surveying); Image (mathematics); Image manipulation; Face (sociological concept); Image compression; Pattern recognition (psychology); Data compression; Image editing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0006871196,0.0003151567,0.0002373153,0.001396593,0.0001797166,0.0003178411,0.0002881171,0.0001898844,0.0001197507],"category_scores_gemma":[0.00009826844,0.000307041,0.00005384923,0.0007285784,0.0002819255,0.0003716378,0.0001459609,0.0002132112,0.0005248929],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002776202,"about_ca_system_score_gemma":0.0003038947,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002238268,"about_ca_topic_score_gemma":0.00169013,"domain_scores_codex":[0.9978284,0.00001133345,0.0002382494,0.000930707,0.0005298994,0.0004613671],"domain_scores_gemma":[0.9991875,0.0000250447,0.0001726274,0.0003478039,0.00008246633,0.00018457],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009954631,0.0001396089,0.00001516813,0.001988834,0.0001910489,0.00001925854,0.002465001,0.0123425,0.1522691,0.04605117,0.5222417,0.2621771],"study_design_scores_gemma":[0.0002444433,0.00004317059,0.000006523436,0.0002239753,0.00007164462,0.00001283233,0.00006301444,0.971944,0.0002391426,0.01279803,0.01401076,0.0003424587],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.00005289962,0.001360763,0.8825723,0.0001687981,0.0007086258,0.001058213,0.0001067986,0.0008347331,0.1131369],"genre_scores_gemma":[0.1434495,0.0001176305,0.2499392,0.00007269302,0.001543555,0.00009970186,0.00010911,0.00203984,0.6026287],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9596015,"threshold_uncertainty_score":0.9999382,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0357763259560234,"score_gpt":0.314082109441217,"score_spread":0.2783057834851935,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}