{"id":"W4410782593","doi":"10.3390/ai6060109","title":"What We Know About the Role of Large Language Models for Medical Synthetic Dataset Generation","year":2025,"lang":"en","type":"article","venue":"AI","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Fundação para a Ciência e a Tecnologia","keywords":"Computer science; Natural language processing; Data science","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05358166,0.001012699,0.001548762,0.002831526,0.0006669922,0.007449004,0.002564188,0.002077919,0.004315807],"category_scores_gemma":[0.2248468,0.000795504,0.002603177,0.002053554,0.002169275,0.00949648,0.002337084,0.002464577,0.002036847],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001815316,"about_ca_system_score_gemma":0.00681671,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004135698,"about_ca_topic_score_gemma":0.005261399,"domain_scores_codex":[0.9640965,0.02939061,0.001960651,0.002020624,0.002256327,0.0002751877],"domain_scores_gemma":[0.606679,0.3616172,0.005247135,0.01490449,0.01043367,0.001118526],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005415663,0.0002343228,0.01874307,0.02956608,0.001940183,0.0002082029,0.00219658,0.04674262,0.002705301,0.03925375,0.02742383,0.8304446],"study_design_scores_gemma":[0.0004323185,0.001580529,0.01744013,0.04949944,0.003168113,0.001576943,0.003717994,0.2278116,0.01433613,0.2532584,0.4266034,0.0005750411],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.04522935,0.4572994,0.4189139,0.05038029,0.001452402,0.0008814603,0.008237363,0.003350405,0.01425545],"genre_scores_gemma":[0.3255899,0.1595169,0.4869961,0.01024207,0.001678063,0.002648085,0.01064532,0.001361164,0.001322505],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.05358166,"threshold_uncertainty_score":0.2833703,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01422989031921169,"score_gpt":0.3408195950236957,"score_spread":0.326589704704484,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}