{"id":"W4416144301","doi":"10.1145/3721201.3721368","title":"Explainable Feature Engineering in Health Data Science: Empirical Comparison of ChatGPT-4o and Classical Machine Learning Methods","year":2025,"lang":"","type":"article","venue":"","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Interpretability; Feature selection; Robustness (evolution); Feature (linguistics); Feature engineering; Ranking (information retrieval); Health care; Feature extraction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02272743,0.0007185709,0.0008694502,0.002413682,0.0007227944,0.001898851,0.001474478,0.001184112,0.001895487],"category_scores_gemma":[0.1320585,0.0002539294,0.0014636,0.002936926,0.001638278,0.003638418,0.002886336,0.002114777,0.0004021544],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001258629,"about_ca_system_score_gemma":0.001761454,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002506269,"about_ca_topic_score_gemma":0.002506483,"domain_scores_codex":[0.9856646,0.01054807,0.0006477833,0.001279762,0.0016479,0.0002118784],"domain_scores_gemma":[0.7687154,0.2115657,0.005230569,0.00936256,0.004085167,0.001040568],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002735635,0.000987744,0.2714817,0.003302704,0.002007264,0.0007019139,0.002643408,0.1478747,0.00210859,0.03516994,0.01520551,0.5157809],"study_design_scores_gemma":[0.0002494508,0.001132725,0.07707421,0.000488229,0.0005116997,0.0008455389,0.001302112,0.8526646,0.00213917,0.05240302,0.01105396,0.0001352378],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5473422,0.008594499,0.4294012,0.00301873,0.00025419,0.0008390161,0.003462413,0.001623167,0.005464669],"genre_scores_gemma":[0.9033916,0.0007572101,0.08962621,0.0003818678,0.0001349318,0.0005020284,0.004317407,0.0001890584,0.0006996997],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02272743,"threshold_uncertainty_score":0.1201956,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3062673893853444,"score_gpt":0.5870522462114316,"score_spread":0.2807848568260872,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}