{"id":"W4412163761","doi":"10.1158/1557-3265.aimachine-b018","title":"Abstract B018: Practical benchmarking of large language models for structuring synthetic prostate cancer histopathology statements in English and Finnish","year":2025,"lang":"en","type":"article","venue":"Clinical Cancer Research","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Prostate cancer; Histopathology; Benchmarking; Medicine; Cancer; Structuring; Pathology; Medical physics; Internal medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004488909,0.001676167,0.0005034131,0.0008188085,0.0005112774,0.001132565,0.001755196,0.001772993,0.005953151],"category_scores_gemma":[0.01826774,0.0004784694,0.001034323,0.0006309764,0.0006558847,0.001428399,0.001097941,0.001300756,0.002544219],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001713179,"about_ca_system_score_gemma":0.001356645,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02113212,"about_ca_topic_score_gemma":0.01553164,"domain_scores_codex":[0.9979417,0.001044067,0.0001517491,0.000502813,0.0002409043,0.0001187686],"domain_scores_gemma":[0.9848262,0.0123912,0.0002493734,0.0008081004,0.001392482,0.0003325849],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002983094,0.001117284,0.01507904,0.001531027,0.0004931387,0.001284543,0.00179874,0.7154257,0.0105468,0.003895058,0.0475224,0.1983231],"study_design_scores_gemma":[0.0002909533,0.0004422475,0.002831033,0.0000744825,0.00006920646,0.0001631976,0.0005338864,0.978609,0.008991047,0.001939992,0.005988949,0.00006611622],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.8459217,0.001254052,0.08816916,0.001309801,0.0005381732,0.0007313273,0.01602189,0.03330255,0.01275125],"genre_scores_gemma":[0.8859777,0.0002974362,0.08310977,0.0003395079,0.00006049198,0.0004831618,0.02502307,0.001300651,0.003408137],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.02113212,"threshold_uncertainty_score":0.04201823,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09570608012282598,"score_gpt":0.5610822752971152,"score_spread":0.4653761951742892,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}