{"id":"W4313457532","doi":"10.21203/rs.3.rs-2393890/v1","title":"Impact of harmonization and oversampling methods on radiomics analysis of multi-center imbalanced datasets: Application to PET-based prediction of lung cancer subtypes","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":8,"is_retracted":false,"has_abstract":false,"ca_institutions":"Occupational Cancer Research Centre","funders":"Basic and Applied Basic Research Foundation of Guangdong Province; Science and Technology Planning Project of Guangdong Province; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Oversampling; Logistic regression; Artificial intelligence; Harmonization; Random forest; Support vector machine; Linear discriminant analysis; Computer science; Machine learning; Radiomics; Cutoff; Medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02234935,0.001423461,0.002130309,0.001570427,0.001098062,0.002539166,0.001814009,0.001728559,0.001103936],"category_scores_gemma":[0.02687751,0.0005579359,0.001797994,0.001765958,0.001220728,0.001857928,0.002478851,0.001530538,0.0004244126],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006636138,"about_ca_system_score_gemma":0.002497404,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007892611,"about_ca_topic_score_gemma":0.007802636,"domain_scores_codex":[0.9945912,0.003084281,0.000361078,0.001092628,0.0005641828,0.000306716],"domain_scores_gemma":[0.9880541,0.007132455,0.0006261736,0.002565539,0.001347739,0.0002740527],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005572673,0.00115993,0.05140838,0.0005373465,0.002587548,0.0002604792,0.0008592414,0.382074,0.02126478,0.004231969,0.01096147,0.5190822],"study_design_scores_gemma":[0.0002346866,0.0002855879,0.01073215,0.00004311756,0.0004036959,0.0001349033,0.0002883819,0.974026,0.00742251,0.004101972,0.002276951,0.00005000627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5248443,0.005351213,0.4603077,0.001399458,0.0005853215,0.0003663166,0.001496054,0.004297041,0.001352653],"genre_scores_gemma":[0.7853234,0.0008000117,0.2074153,0.0004886189,0.0003243818,0.0001898616,0.003768623,0.0007462588,0.0009435624],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02234935,"threshold_uncertainty_score":0.118196,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0736727508529272,"score_gpt":0.5064170616909489,"score_spread":0.4327443108380217,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}