{"id":"W7131085793","doi":"10.1109/iccvw69036.2025.00072","title":"Are Medical Image Generative Models Biologically Trustworthy?","year":2025,"lang":"","type":"article","venue":"","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Generative grammar; Generative model; Perception; Trustworthiness; Focus (optics); Computational model; Cognition","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001854623,0.0006640667,0.0008711952,0.0003812498,0.0007915049,0.0009542359,0.003856237,0.000740397,0.003608179],"category_scores_gemma":[0.001649578,0.0005443998,0.0003671027,0.002372069,0.001046114,0.001637166,0.002352598,0.000936417,0.001128313],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002279295,"about_ca_system_score_gemma":0.001181931,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002462865,"about_ca_topic_score_gemma":0.0005407712,"domain_scores_codex":[0.9933641,0.000750266,0.00140011,0.001928218,0.00113744,0.001419869],"domain_scores_gemma":[0.9956878,0.0006698011,0.0004363226,0.001571947,0.0009853775,0.0006487976],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000449285,0.000504192,0.0002495217,0.0000409277,0.0001218392,0.0005967739,0.0006851372,0.00289345,0.0007321328,0.8762676,0.01241775,0.1054458],"study_design_scores_gemma":[0.0001919172,0.0001203766,0.0001646487,0.0001830854,0.00002240965,0.00001306622,0.0006631149,0.8036273,0.01807486,0.1736363,0.00277139,0.0005315681],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006864446,0.001201214,0.8658531,0.02760486,0.001766056,0.0005789157,0.000009023735,0.0003314168,0.09579101],"genre_scores_gemma":[0.8904127,0.001097158,0.06896835,0.02105984,0.0004544212,0.0001177677,0.00000529342,0.00002893452,0.01785556],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8835483,"threshold_uncertainty_score":0.9997007,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05275318530772239,"score_gpt":0.3201313267330086,"score_spread":0.2673781414252862,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}