{"id":"W4376473362","doi":"10.1615/critrevoncog.2023048797","title":"Perspective of a Pathologist on Benchmark Strategies for Artificial Intelligence Development in Organ Transplantation","year":2023,"lang":"en","type":"review","venue":"Critical Reviews™ in Oncogenesis","topic":"Prenatal Screening and Diagnostics","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University Hospital Foundation","funders":"","keywords":"Interpretability; Digital pathology; Transplantation; Medicine; Process (computing); Computer science; Gold standard (test); Intensive care medicine; Artificial intelligence; Medical physics; Pathology; Surgery; Radiology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001209957,0.000361202,0.002172099,0.000379876,0.00003858381,0.00001927091,0.0001796702,0.0003216803,0.0000243888],"category_scores_gemma":[0.009635763,0.0002826134,0.0003423613,0.0007601418,0.0001685297,0.00004238976,0.00003304829,0.0003163428,0.00006540473],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004057494,"about_ca_system_score_gemma":0.0007978996,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005838239,"about_ca_topic_score_gemma":0.0002036735,"domain_scores_codex":[0.9970322,0.0002589322,0.0015342,0.000537021,0.0002483185,0.0003892937],"domain_scores_gemma":[0.9956981,0.003612398,0.0001574551,0.0002556921,0.0001600071,0.0001163472],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.00008221609,0.0002926423,0.0000108564,0.05599656,0.00002258804,0.0002044019,0.0003138295,0.000002058583,0.000001769592,0.03337516,0.00002735035,0.9096706],"study_design_scores_gemma":[0.0007278087,0.002361142,0.0004798239,0.6542961,0.003972825,0.0001800854,0.003346293,0.00007013257,0.0004756459,0.02585474,0.3062876,0.001947778],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.00004894683,0.9917858,0.005231388,0.0001187485,0.0001306429,0.00202684,0.0001328292,0.00002842834,0.0004964017],"genre_scores_gemma":[0.0007696345,0.9935925,0.004419054,0.00004715,0.00008906028,0.0008133178,0.0002152897,0.00004179675,0.00001222594],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9077228,"threshold_uncertainty_score":0.9999626,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2538991459600352,"score_gpt":0.478408049979496,"score_spread":0.2245089040194608,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}