{"id":"W4392356050","doi":"10.29115/sp-2023-0022","title":"Keep the noise down: On the performance of automatic speech recognition of voice-recordings in web surveys","year":2024,"lang":"en","type":"article","venue":"Survey Practice","topic":"Survey Methodology and Nonresponse","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Speech recognition; Noise (video); Computer science; Voice activity detection; Speech processing; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3390485,0.0001226353,0.000234132,0.0001518553,0.000245447,0.0000515971,0.0003990425,0.0001467572,0.0003866426],"category_scores_gemma":[0.1658369,0.00007752945,0.00007337341,0.001457451,0.0004477557,0.0005653224,0.00004727755,0.0005352163,0.0002166519],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006026453,"about_ca_system_score_gemma":0.0004662189,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0111156,"about_ca_topic_score_gemma":0.01786322,"domain_scores_codex":[0.8278798,0.1702576,0.000580635,0.0003147369,0.0006156352,0.0003516047],"domain_scores_gemma":[0.791437,0.2078058,0.0002299729,0.0002581365,0.0002394265,0.00002969823],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01313659,0.001012532,0.20336,0.0004304906,0.0004897892,0.00006817243,0.04797045,0.00004877051,0.003697724,0.001062006,0.005581076,0.7231424],"study_design_scores_gemma":[0.0002319615,0.0002234889,0.9837419,0.0002462018,0.00005536801,0.000008545118,0.004495723,0.001302358,0.003110574,0.0003095251,0.006091415,0.0001829396],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9854515,0.0002291699,0.00002756796,0.003441511,0.0008919215,0.0003643835,0.00005218307,0.00004122495,0.009500498],"genre_scores_gemma":[0.9981439,0.0005679359,0.0004602039,0.0002135365,0.00007044467,0.00002293756,0.0000102621,0.00001317678,0.0004976077],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7803819,"threshold_uncertainty_score":0.9968098,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2129304017765198,"score_gpt":0.426281586716043,"score_spread":0.2133511849395232,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}