{"id":"W4407241355","doi":"10.36227/techrxiv.173894988.81604432/v1","title":"Towards a Unified Evaluation Framework: Integrating Human Perception and Metrics for AI-Generated Images","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Advanced Neural Network Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Perception; Computer science; Artificial intelligence; Data science; Computer vision; Psychology; Neuroscience","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02053357,0.002050003,0.001024029,0.005109961,0.0007314525,0.005826049,0.002305686,0.001577602,0.00275736],"category_scores_gemma":[0.06557164,0.0004071044,0.000741971,0.001948299,0.00269731,0.006172194,0.003782522,0.002178889,0.000927021],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00249587,"about_ca_system_score_gemma":0.002135152,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004791919,"about_ca_topic_score_gemma":0.005049425,"domain_scores_codex":[0.9833655,0.008950246,0.001004062,0.001442145,0.004865236,0.0003728232],"domain_scores_gemma":[0.9621541,0.01534673,0.003082259,0.004120439,0.01380718,0.001489259],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0007373114,0.0008845611,0.01547681,0.001908992,0.0003213055,0.0001763348,0.002964151,0.06937099,0.04256038,0.1062068,0.01372202,0.7456702],"study_design_scores_gemma":[0.00009519175,0.001555109,0.01459028,0.0009005569,0.0001808568,0.0003376772,0.002315191,0.7906199,0.03414065,0.1321182,0.022843,0.0003034354],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02593932,0.0009429387,0.9611164,0.0009947895,0.0001144908,0.000550439,0.0002549602,0.001932947,0.008153707],"genre_scores_gemma":[0.3845063,0.0004810115,0.6115096,0.0002797116,0.00008432226,0.0006627594,0.0006090194,0.0004480673,0.001419138],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9794664,"threshold_uncertainty_score":0.1085932,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07612113581608967,"score_gpt":0.3963464464994623,"score_spread":0.3202253106833726,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}