{"id":"W4416037063","doi":"10.18653/v1/2025.emnlp-main.136","title":"SurveyGen: Quality-Aware Scientific Survey Generation with Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Survey Methodology and Nonresponse","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; University of Alberta","keywords":"Natural language generation; Natural language; Language model; Government (linguistics)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","sts","insufficient_payload"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2604465,0.0003907922,0.000657479,0.0004336451,0.003137105,0.0006589966,0.0007208319,0.0005729509,0.002656423],"category_scores_gemma":[0.01178767,0.000339523,0.0001531652,0.003597741,0.001274012,0.0007343585,0.0001959049,0.0004495973,0.000132728],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001611363,"about_ca_system_score_gemma":0.002435168,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.02221568,"about_ca_topic_score_gemma":0.4003247,"domain_scores_codex":[0.8363769,0.1590724,0.0008943306,0.001453391,0.0009785971,0.001224309],"domain_scores_gemma":[0.9778809,0.01972966,0.0002561302,0.0009339486,0.0009624609,0.0002369273],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.03532832,0.004157291,0.4499782,0.0003514773,0.001900298,0.0001650927,0.1940982,0.004429584,0.006087547,0.2333694,0.05292769,0.01720694],"study_design_scores_gemma":[0.004641197,0.0003596602,0.8922521,0.0001617532,0.0002307116,0.000004563695,0.04663811,0.03520566,0.008325221,0.002662006,0.007143753,0.00237525],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7619435,0.001260204,0.2152409,0.0006952867,0.003567067,0.0006433623,0.0005144486,0.0001271027,0.01600816],"genre_scores_gemma":[0.8944214,0.00007675464,0.0009633853,0.0005551603,0.0001564217,0.00002537936,0.000401755,0.00001996818,0.1033798],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4422739,"threshold_uncertainty_score":0.9999057,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3905043516417337,"score_gpt":0.498332670281995,"score_spread":0.1078283186402612,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}