{"id":"W4210506170","doi":"10.1002/mp.15514","title":"On the proper use of structural similarity for the robust evaluation of medical image synthesis models","year":2022,"lang":"en","type":"article","venue":"Medical Physics","topic":"Image and Video Quality Assessment","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Normalization (sociology); Voxel; Image quality; Artificial intelligence; Metric (unit); Pattern recognition (psychology); Computer science; Medical imaging; Similarity (geometry); Mathematics; Computation; Image processing; Image (mathematics); Algorithm","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01201947,0.00182484,0.001016921,0.002360844,0.0005735742,0.00240348,0.001511239,0.002057063,0.001666632],"category_scores_gemma":[0.0615507,0.0004693141,0.000717462,0.0009406884,0.002068261,0.002645749,0.002795662,0.001749788,0.0006890719],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00101741,"about_ca_system_score_gemma":0.001600806,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002029883,"about_ca_topic_score_gemma":0.002133989,"domain_scores_codex":[0.9931699,0.002905245,0.0007408017,0.0007634143,0.002247763,0.0001728688],"domain_scores_gemma":[0.9782601,0.01242184,0.00212938,0.002240262,0.004641183,0.0003072742],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006127388,0.0003012441,0.007118684,0.001108053,0.0003497052,0.0002507523,0.0003207238,0.3581873,0.05238813,0.0208657,0.003226958,0.55527],"study_design_scores_gemma":[0.00002800011,0.0004804506,0.002010432,0.0003303122,0.0000603227,0.0003008927,0.00008872927,0.9476601,0.03439239,0.01181713,0.002776755,0.00005431242],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03457251,0.002952368,0.9578245,0.0008031766,0.0001203118,0.0002001359,0.0001218213,0.0009886744,0.002416398],"genre_scores_gemma":[0.4844288,0.001867325,0.5107704,0.0004520927,0.0001765725,0.0002995338,0.0005721611,0.000445737,0.00098741],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9879805,"threshold_uncertainty_score":0.06356579,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2245681303530576,"score_gpt":0.3715333982732142,"score_spread":0.1469652679201567,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}