{"id":"W4400255094","doi":"10.20944/preprints202406.2082.v1","title":"TIPAA-SSL: Text Independent Phone-to-Audio Alignment based on Self-Supervised Learning and Knowledge Transfer","year":2024,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Phone; Speech recognition; Artificial intelligence; TIMIT; Classifier (UML); Transfer of learning; Natural language processing; Frame (networking); Hidden Markov model","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001296874,0.001507522,0.001236013,0.001271097,0.0006915478,0.001286585,0.003248871,0.001972675,0.008280744],"category_scores_gemma":[0.003839065,0.0006153584,0.00110913,0.001171577,0.0007697128,0.003313257,0.002675002,0.002547124,0.0114736],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006123637,"about_ca_system_score_gemma":0.001315299,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00337525,"about_ca_topic_score_gemma":0.005785283,"domain_scores_codex":[0.9984529,0.0003446806,0.00007603235,0.0006146409,0.0003812197,0.0001304347],"domain_scores_gemma":[0.9982386,0.0004552624,0.0001183966,0.0006400535,0.0004423499,0.0001053694],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004087636,0.0004836836,0.0009661736,0.0001869232,0.0001493674,0.0001832137,0.000140625,0.05137184,0.02307632,0.003395574,0.02889697,0.8907405],"study_design_scores_gemma":[0.00004481101,0.0001533175,0.0005236886,0.00001670218,0.00003023561,0.0001201167,0.00004326906,0.9644867,0.02008599,0.007232197,0.007229512,0.00003345743],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01260348,0.0004617416,0.9499933,0.0001914195,0.0003366968,0.0001572263,0.0007972734,0.03211641,0.00334246],"genre_scores_gemma":[0.2922313,0.0003313209,0.6693246,0.0008709751,0.00045576,0.000668421,0.008650369,0.002661614,0.02480559],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008280744,"threshold_uncertainty_score":0.02770185,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07535121666241719,"score_gpt":0.3170469676804163,"score_spread":0.2416957510179991,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}