{"id":"W4388624949","doi":"10.2139/ssrn.4631449","title":"Combining Survey and Census Data for Improved Poverty Prediction Using Semi-Supervised Deep Learning","year":2023,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Human Mobility and Location-Based Analysis","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Census; Poverty; Survey data collection; Deep learning; American Community Survey; Artificial intelligence; Machine learning; Statistics; Geography; Computer science; Econometrics; Mathematics; Demography; Economic growth; Economics; Sociology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.01688715,0.0002325697,0.0004108366,0.0002046621,0.002067326,0.0004108734,0.0007456348,0.000355209,0.00001648007],"category_scores_gemma":[0.002419671,0.0002442958,0.000147734,0.0002990362,0.0001421368,0.0002523427,0.0003752675,0.003185622,0.000001726894],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001422337,"about_ca_system_score_gemma":0.00460827,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.02715339,"about_ca_topic_score_gemma":0.2006457,"domain_scores_codex":[0.9956368,0.001122019,0.0005284722,0.0006577809,0.0003719467,0.001682969],"domain_scores_gemma":[0.9979199,0.0006641761,0.0004348961,0.000447151,0.0003789246,0.0001549811],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001484921,0.0009020343,0.364429,0.0009831219,0.01019852,0.000009587486,0.03410963,0.1196927,0.001515844,0.02125295,0.001193992,0.4442276],"study_design_scores_gemma":[0.001094737,0.0001778594,0.005086282,0.0001472869,0.0005449981,0.000008682044,0.0181582,0.9078425,0.000002619704,0.06449141,0.001905678,0.0005398059],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.517422,0.006368132,0.4678482,0.002505695,0.002134882,0.001842527,0.001236626,0.0004665872,0.0001754284],"genre_scores_gemma":[0.9891192,0.006715187,0.0002346859,0.00004988867,0.0008545609,0.00001491045,0.002230617,0.00005374199,0.0007271793],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7881497,"threshold_uncertainty_score":0.9992319,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08075252628813522,"score_gpt":0.3423030498362135,"score_spread":0.2615505235480782,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}