{"id":"W4400255094","doi":"10.20944/preprints202406.2082.v1","title":"TIPAA-SSL: Text Independent Phone-to-Audio Alignment based on Self-Supervised Learning and Knowledge Transfer","year":2024,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Phone; Speech recognition; Artificial intelligence; TIMIT; Classifier (UML); Transfer of learning; Natural language processing; Frame (networking); Hidden Markov model","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001471654,0.0006722818,0.0006522934,0.0006340892,0.0001981185,0.000282196,0.001335451,0.0004608037,0.001083456],"category_scores_gemma":[0.0001557914,0.0006683466,0.0003539706,0.0004191588,0.00004900708,0.0001043418,0.003089166,0.001597076,0.009298033],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003531782,"about_ca_system_score_gemma":0.0004042672,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006074499,"about_ca_topic_score_gemma":0.0000234379,"domain_scores_codex":[0.995143,0.000505329,0.0006489402,0.002294234,0.0008118338,0.0005966588],"domain_scores_gemma":[0.9975889,0.0003141291,0.00008389325,0.001357343,0.0001615907,0.0004941836],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005863565,0.00572108,0.1056766,0.005389374,0.002498315,0.0009918304,0.05392666,0.005610502,0.04300193,0.009059978,0.002032532,0.7655049],"study_design_scores_gemma":[0.00424887,0.000445295,0.07436269,0.004555157,0.0006108448,0.0001038098,0.0006037394,0.4224735,0.3874874,0.008618666,0.09097575,0.005514345],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8447241,0.0003838761,0.03236135,0.002976752,0.002504504,0.002089064,0.00003218158,0.002221355,0.1127068],"genre_scores_gemma":[0.9899939,0.0001536854,0.006169863,0.0006538485,0.0001710293,0.000480701,0.00001275752,0.00008206261,0.00228219],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7599905,"threshold_uncertainty_score":0.9998297,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07535121666241719,"score_gpt":0.3170469676804163,"score_spread":0.2416957510179991,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}