{"id":"W4409186862","doi":"10.1007/s00530-025-01769-7","title":"Towards a unified evaluation framework: integrating human perception and metrics for AI-generated images","year":2025,"lang":"en","type":"article","venue":"Multimedia Systems","topic":"Visual Attention and Saliency Detection","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Perception; Computer graphics; Cryptography; Artificial intelligence; Data science; Human–computer interaction; Computer security; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01595725,0.001696273,0.001440896,0.00446799,0.0007134852,0.005411834,0.002638233,0.001774666,0.002120414],"category_scores_gemma":[0.04267723,0.0004226732,0.0007564952,0.001807804,0.001732356,0.005184516,0.002912529,0.001999238,0.0006957491],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002405126,"about_ca_system_score_gemma":0.002373489,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007636398,"about_ca_topic_score_gemma":0.007776094,"domain_scores_codex":[0.9897065,0.004717581,0.0008340544,0.001162468,0.003260807,0.0003186522],"domain_scores_gemma":[0.976117,0.009416498,0.001869828,0.002334443,0.008898797,0.001363423],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0009491001,0.0007877123,0.01625379,0.001495732,0.000602967,0.0001588453,0.001144202,0.07208175,0.04298564,0.08429296,0.01416402,0.7650832],"study_design_scores_gemma":[0.00005461607,0.0007424849,0.00810698,0.0002306793,0.0001822347,0.0002036989,0.0004623391,0.8820639,0.01886622,0.08028991,0.008692702,0.0001043204],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01747049,0.001057808,0.9755636,0.0006132079,0.00007264943,0.0002938923,0.0002537538,0.001657177,0.003017501],"genre_scores_gemma":[0.4379602,0.0005250683,0.5581007,0.0002493103,0.0001366725,0.0003725972,0.000662299,0.0004933515,0.001499814],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9840428,"threshold_uncertainty_score":0.084391,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05179110395672034,"score_gpt":0.3740683480577116,"score_spread":0.3222772441009912,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}