{"id":"W3002402144","doi":"10.1088/1361-6560/ab6e54","title":"Data clustering to select clinically-relevant test cases for algorithm benchmarking and characterization","year":2020,"lang":"en","type":"article","venue":"Physics in Medicine and Biology","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Lethbridge; University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Medoid; Benchmarking; Cluster analysis; Algorithm; Computer science; Workflow; Data mining; Ground truth; Statistical hypothesis testing; Artificial intelligence; Statistics; Mathematics; Database","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004506525,0.0001164177,0.0003952131,0.00004281609,0.00005007183,0.000008156939,0.00009090531,0.00005659866,0.000005906351],"category_scores_gemma":[0.002146205,0.0000867195,0.00001331556,0.0001455363,0.0001220223,0.00004298409,0.0001680436,0.0002411958,6.315617e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00000871827,"about_ca_system_score_gemma":0.00002832047,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003147591,"about_ca_topic_score_gemma":0.000003683663,"domain_scores_codex":[0.999003,0.00003273995,0.0003131532,0.0004012898,0.00005633081,0.0001935004],"domain_scores_gemma":[0.9988045,0.0007612127,0.00007244472,0.0001446696,0.00003808002,0.0001790892],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001035098,0.00004583546,0.04663211,0.0001853576,0.00002909731,0.00002361503,0.0005228778,0.00000377277,0.05107417,0.0000970149,0.0005121615,0.9007705],"study_design_scores_gemma":[0.003544849,0.004196039,0.0196803,0.000535761,0.0001546783,0.0001277631,0.0001785571,0.9278455,0.00008349717,0.000435395,0.04296223,0.0002553993],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.356842,0.0004256789,0.573612,0.06761064,0.0003386885,0.0008694789,0.0001315769,0.0000592277,0.0001107427],"genre_scores_gemma":[0.9289647,0.001384467,0.03911391,0.02499769,0.0040813,0.00002426499,0.001389361,0.00003006696,0.00001424956],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9278418,"threshold_uncertainty_score":0.3536319,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2363964357711582,"score_gpt":0.4405806988614017,"score_spread":0.2041842630902435,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}