{"id":"W4407635325","doi":"10.1111/his.15432","title":"Comparative performance of <scp>PD</scp>‐<scp>L1</scp> scoring by pathologists and <scp>AI</scp> algorithms","year":2025,"lang":"en","type":"article","venue":"Histopathology","topic":"Cancer Immunotherapy and Biomarkers","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; BC Cancer Agency","funders":"HORIZON EUROPE Framework Programme; Austrian Science Fund; European Commission","keywords":"Concordance; Kappa; Medicine; Cohen's kappa; Lung cancer; Algorithm; Nuclear medicine; Oncology; Internal medicine; Machine learning; Computer science; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03050425,0.0008254168,0.0007678128,0.003865706,0.0005421056,0.002118643,0.0007310602,0.001213649,0.001102545],"category_scores_gemma":[0.05035496,0.000373289,0.000894176,0.001039218,0.0009680861,0.001273633,0.001540113,0.0005298071,0.0005949892],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007165847,"about_ca_system_score_gemma":0.0005022967,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007639647,"about_ca_topic_score_gemma":0.000999118,"domain_scores_codex":[0.9753885,0.013205,0.002124689,0.004034151,0.004736058,0.0005116204],"domain_scores_gemma":[0.9524476,0.03152349,0.00469458,0.003018277,0.007240132,0.001075949],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.01053454,0.0003574172,0.698961,0.0009227115,0.00219907,0.0002179048,0.002292819,0.009726875,0.02002122,0.0006104476,0.001249234,0.2529068],"study_design_scores_gemma":[0.0003694761,0.006743763,0.839696,0.0003834052,0.001568497,0.002450261,0.002158441,0.1034109,0.03570748,0.001843169,0.005437308,0.0002312951],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9737371,0.003222408,0.01690239,0.0001772327,0.00010853,0.0002279862,0.0003154911,0.0003590337,0.00494971],"genre_scores_gemma":[0.9824765,0.0003883144,0.01582377,0.00006451389,0.00005134763,0.00008600192,0.0005039839,0.0000695765,0.0005360423],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03050425,"threshold_uncertainty_score":0.1613238,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0181233056759116,"score_gpt":0.2860200487901736,"score_spread":0.267896743114262,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}