{"id":"W4402191750","doi":"10.1080/07268602.2024.2368780","title":"Building a searchable online corpus of Australian and New Zealand aligned speech","year":2024,"lang":"en","type":"article","venue":"Australian Journal of Linguistics","topic":"Linguistic Variation and Morphology","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Python (programming language); Upload; World Wide Web; Login; Software; The Internet","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003117737,0.001136392,0.001039472,0.007002847,0.002548087,0.001841436,0.002194538,0.001076789,0.02800069],"category_scores_gemma":[0.01436075,0.001003089,0.0006455733,0.007898779,0.001315319,0.002689838,0.004956859,0.001757203,0.0150979],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00250362,"about_ca_system_score_gemma":0.007321957,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.20709,"about_ca_topic_score_gemma":0.2469261,"domain_scores_codex":[0.9964174,0.0005263183,0.0004821951,0.001134874,0.001163949,0.0002752587],"domain_scores_gemma":[0.9901325,0.001951224,0.0007749201,0.001768577,0.004636985,0.0007357789],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001004271,0.0006358618,0.0138583,0.006709922,0.0002293427,0.005557884,0.03372613,0.003017524,0.08661749,0.01035524,0.4030549,0.4352332],"study_design_scores_gemma":[0.0002489301,0.0002948406,0.2260573,0.0006169775,0.0002594461,0.002709472,0.009026159,0.008240193,0.01478046,0.002853384,0.7345467,0.0003662289],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.3407986,0.002966946,0.06420045,0.001778574,0.001123087,0.006713477,0.4919988,0.005834367,0.08458582],"genre_scores_gemma":[0.2740299,0.001766366,0.1118243,0.000513299,0.0002699538,0.009729321,0.5571325,0.00237148,0.04236279],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.20709,"threshold_uncertainty_score":0.4117692,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07384442457558994,"score_gpt":0.3845869942070119,"score_spread":0.310742569631422,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}