{"id":"W4313457532","doi":"10.21203/rs.3.rs-2393890/v1","title":"Impact of harmonization and oversampling methods on radiomics analysis of multi-center imbalanced datasets: Application to PET-based prediction of lung cancer subtypes","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":8,"is_retracted":false,"has_abstract":false,"ca_institutions":"Occupational Cancer Research Centre","funders":"Basic and Applied Basic Research Foundation of Guangdong Province; Science and Technology Planning Project of Guangdong Province; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Oversampling; Logistic regression; Artificial intelligence; Harmonization; Random forest; Support vector machine; Linear discriminant analysis; Computer science; Machine learning; Radiomics; Cutoff; Medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002346031,0.0001734066,0.0007392688,0.001638605,0.00005093811,0.00001845786,0.0001775068,0.0001322321,0.00002075666],"category_scores_gemma":[0.001833006,0.0001502069,0.000227412,0.001391882,0.0001323381,0.00003191313,0.0002125987,0.0007168308,3.541405e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003798375,"about_ca_system_score_gemma":0.0003498459,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002778202,"about_ca_topic_score_gemma":0.00002689098,"domain_scores_codex":[0.9976282,0.0003536203,0.000547965,0.0005088988,0.0007070584,0.0002542356],"domain_scores_gemma":[0.9976597,0.00059932,0.0003369476,0.0006366491,0.0005824408,0.0001848807],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007350096,0.0004388984,0.7034712,0.002784908,0.002034928,0.000003608789,0.0004125724,0.245532,0.0329147,0.00001863067,0.0004782983,0.0111752],"study_design_scores_gemma":[0.0008926188,0.0001513266,0.3557244,0.0009340888,0.0004386315,3.772998e-7,0.00003048392,0.6403751,0.001365911,0.000005913168,0.00002403882,0.00005705254],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6631137,0.0003311328,0.3306741,0.000361426,0.00007457726,0.001160493,0.004245829,0.00003184897,0.000006837576],"genre_scores_gemma":[0.9697693,0.0008450232,0.02251879,0.00001774961,0.00006584565,0.0001085543,0.006618954,0.00004302367,0.00001275937],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3948431,"threshold_uncertainty_score":0.6125259,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0736727508529272,"score_gpt":0.5064170616909489,"score_spread":0.4327443108380217,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}