{"id":"W4389387146","doi":"10.3233/sji-230063","title":"Classifying respondent comments from the 2021 Canadian Census of Population using machine learning methods1","year":2023,"lang":"en","type":"article","venue":"Statistical Journal of the IAOS","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Respondent; Census; Computer science; Categorization; Encoder; Population; Artificial intelligence; Transformer; Machine learning; Natural language processing; Statistics; Econometrics; Geography; Demography; Mathematics; Sociology; Engineering; Political science","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006861367,0.0007208467,0.0003344322,0.00439783,0.001392049,0.0009892804,0.0008923771,0.0004033365,0.003775087],"category_scores_gemma":[0.02839313,0.0001902571,0.0004026518,0.004971301,0.0004269146,0.0005260746,0.0009744518,0.0007088748,0.001876991],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007764371,"about_ca_system_score_gemma":0.01531091,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.7717856,"about_ca_topic_score_gemma":0.8593895,"domain_scores_codex":[0.9962606,0.0007879216,0.0002373179,0.0002924455,0.002102459,0.0003192548],"domain_scores_gemma":[0.9703976,0.00696904,0.001259006,0.001097882,0.01950043,0.0007760178],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0007947118,0.0004190649,0.3859758,0.001798917,0.000148967,0.0003956798,0.02068552,0.008246616,0.01301474,0.002889522,0.1358441,0.4297864],"study_design_scores_gemma":[0.00009974439,0.0003092087,0.6709774,0.0005384053,0.0001178086,0.0001546532,0.03052572,0.07278279,0.01890719,0.001854644,0.20344,0.0002924189],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7807913,0.0007230596,0.04488508,0.002673218,0.0005220841,0.004480274,0.1366223,0.002620124,0.02668261],"genre_scores_gemma":[0.7827197,0.0009136779,0.09626047,0.0006255868,0.0001703096,0.003328754,0.08965196,0.0004057469,0.02592378],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2282144,"threshold_uncertainty_score":0.4591169,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09543310328496614,"score_gpt":0.3544916148562567,"score_spread":0.2590585115712906,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}