{"id":"W4412151029","doi":"10.1080/14664208.2025.2524285","title":"Evaluating the validity of census data for tracking speaker numbers: an investigation of Canada’s Indigenous languages","year":2025,"lang":"en","type":"article","venue":"Current Issues in Language Planning","topic":"Linguistic Variation and Morphology","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Wenner-Gren Stiftelserna","keywords":"Census; Indigenous; Tracking (education); Linguistics; Geography; Statistics; Natural language processing; Genealogy; Computer science; Sociology; History; Demography; Mathematics; Population; Pedagogy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03953493,0.000478983,0.0005183183,0.004396419,0.004531814,0.003302452,0.00300794,0.0004932224,0.0009645177],"category_scores_gemma":[0.1956109,0.00047145,0.000622579,0.009808247,0.003164848,0.001758025,0.003037347,0.001057759,0.0002176094],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02846702,"about_ca_system_score_gemma":0.05860622,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.9901664,"about_ca_topic_score_gemma":0.9900919,"domain_scores_codex":[0.9688712,0.00899402,0.001671736,0.002211776,0.01613154,0.002119698],"domain_scores_gemma":[0.8612395,0.03651806,0.01463517,0.008699051,0.07617506,0.002733123],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00008708394,0.00002057983,0.9518177,0.0001502939,0.0002124261,0.00008794935,0.01523722,0.001832442,0.00018539,0.002786191,0.003817283,0.02376543],"study_design_scores_gemma":[0.00001068809,0.00003153235,0.96772,0.0003406453,0.00009369949,0.00005586159,0.0121931,0.007746741,0.0005200275,0.0006462084,0.01057984,0.00006164873],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9525877,0.001875209,0.01246456,0.004729251,0.0001339845,0.0003471661,0.009963124,0.0001201081,0.01777887],"genre_scores_gemma":[0.9890853,0.0005132292,0.006026218,0.0002917222,0.00002204051,0.0001271722,0.0028829,0.00005456426,0.0009969476],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.971533,"threshold_uncertainty_score":0.2090833,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2232075762142746,"score_gpt":0.5056563730349884,"score_spread":0.2824487968207139,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}