{"id":"W4321789104","doi":"10.31219/osf.io/hzwvf","title":"Reproducible Speech Research with the Artificial-Intelligence-Ready PERCEPT Corpora","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Language Development and Disorders","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Memorial University of Newfoundland","funders":"National Institute on Deafness and Other Communication Disorders; McGill University; National Institutes of Health; Syracuse University; National Science Foundation","keywords":"Percept; Computer science; Natural language processing; Speech corpus; Perception; Artificial intelligence; Speech recognition; Psychology; Speech synthesis","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04421977,0.001638836,0.0009934078,0.004673854,0.002916091,0.006372617,0.004393578,0.002274474,0.05898434],"category_scores_gemma":[0.1465479,0.001426771,0.001296145,0.004058404,0.004083671,0.007567151,0.01122554,0.004245476,0.04617936],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001761747,"about_ca_system_score_gemma":0.007661624,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003359764,"about_ca_topic_score_gemma":0.005528418,"domain_scores_codex":[0.9590458,0.01998863,0.006223791,0.006915426,0.007027376,0.0007990135],"domain_scores_gemma":[0.7924628,0.08090604,0.00682902,0.087221,0.02952245,0.003058628],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001967538,0.001045387,0.01348512,0.005944865,0.0003099466,0.001567848,0.01336186,0.004496229,0.03210276,0.03536214,0.4963956,0.3939607],"study_design_scores_gemma":[0.0006659019,0.0005032216,0.02014753,0.00189889,0.0001885212,0.001573414,0.004057629,0.00774896,0.02038092,0.02293558,0.9195689,0.000330572],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"dataset","genre_scores_codex":[0.04889978,0.002070612,0.4768063,0.007113965,0.003750023,0.01048953,0.3156451,0.0483686,0.08685608],"genre_scores_gemma":[0.09214748,0.0007312191,0.5416131,0.002207875,0.0009162869,0.02280289,0.3061393,0.0163572,0.01708471],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.9557802,"threshold_uncertainty_score":0.2338592,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3900993921982348,"score_gpt":0.4535953938095266,"score_spread":0.06349600161129182,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}