{"id":"W4407635325","doi":"10.1111/his.15432","title":"Comparative performance of <scp>PD</scp>‐<scp>L1</scp> scoring by pathologists and <scp>AI</scp> algorithms","year":2025,"lang":"en","type":"article","venue":"Histopathology","topic":"Cancer Immunotherapy and Biomarkers","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; BC Cancer Agency","funders":"HORIZON EUROPE Framework Programme; Austrian Science Fund; European Commission","keywords":"Concordance; Kappa; Medicine; Cohen's kappa; Lung cancer; Algorithm; Nuclear medicine; Oncology; Internal medicine; Machine learning; Computer science; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.000753511,0.0005049253,0.001230444,0.0004451677,0.0002772545,0.00002873225,0.0003268964,0.0004573985,0.00001267693],"category_scores_gemma":[0.0005490857,0.0004737163,0.0002075515,0.0006230389,0.001319586,0.0001755449,0.0001894276,0.0006772911,0.00003364577],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002954778,"about_ca_system_score_gemma":0.0002852207,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009745389,"about_ca_topic_score_gemma":0.00002457524,"domain_scores_codex":[0.9969645,0.0002347171,0.0008310628,0.0008585289,0.0003274125,0.0007837638],"domain_scores_gemma":[0.9976325,0.000827947,0.0004327271,0.0006781676,0.0002522426,0.0001764163],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.000298435,0.000955231,0.210904,0.001294729,0.0002459606,0.0006379131,0.0154288,0.00002202269,0.5840214,0.0005049809,0.1705294,0.01515714],"study_design_scores_gemma":[0.006384405,0.00256321,0.5492599,0.0005917334,0.0003703624,0.001003862,0.003307326,0.0007130895,0.08888838,0.0001762722,0.3466152,0.0001262504],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9594309,0.02142994,0.0007853115,0.0001829434,0.001201002,0.0006413743,0.00008411245,0.000149367,0.01609503],"genre_scores_gemma":[0.9702826,0.002598633,0.001463939,0.001955536,0.0001690808,0.000129453,0.0001001176,0.00005105904,0.02324963],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.495133,"threshold_uncertainty_score":0.9997715,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0181233056759116,"score_gpt":0.2860200487901736,"score_spread":0.267896743114262,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}