{"id":"W4402671182","doi":"10.18653/v1/2024.acl-srw.34","title":"Homophone2Vec: Embedding Space Analysis for Empirical Evaluation of Phonological and Semantic Similarity","year":2024,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"McGill University","keywords":"Computer science; Semantic similarity; Embedding; Similarity (geometry); Semantic space; Natural language processing; Space (punctuation); Artificial intelligence; Information retrieval","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001094211,0.00007502513,0.0001855634,0.0002528247,0.00004798481,0.0001281735,0.0001394034,0.00006468168,0.0001965559],"category_scores_gemma":[0.0002676105,0.00005692027,0.0001367194,0.0007860612,0.00003374349,0.0001650811,0.00006714489,0.0000507666,0.000009908565],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000251855,"about_ca_system_score_gemma":0.00004857383,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000125656,"about_ca_topic_score_gemma":0.00002362947,"domain_scores_codex":[0.9990108,0.00009268363,0.0001689293,0.0003204624,0.0002879832,0.0001191453],"domain_scores_gemma":[0.9991208,0.0004769208,0.00002761341,0.0001784274,0.0001478455,0.00004839529],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002679584,0.0003260114,0.01524741,0.0002336069,0.001619317,0.00001817595,0.002073283,0.000683648,0.005041123,0.02638155,0.004392329,0.9439567],"study_design_scores_gemma":[0.0001155774,0.00003230175,0.009915048,0.00000996177,0.000316058,0.000005435076,0.00006014495,0.9751368,0.00623974,0.007862408,0.0002268359,0.00007974949],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2420153,0.0002090728,0.7545179,0.001855101,0.00009594432,0.0001459124,0.000003428015,0.0001031159,0.001054319],"genre_scores_gemma":[0.9040338,0.00001670484,0.09572934,0.00008869728,0.0000175611,0.00001860425,0.000002302813,0.000002616985,0.00009032858],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9744531,"threshold_uncertainty_score":0.2321141,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1165663195041532,"score_gpt":0.3838189603201362,"score_spread":0.2672526408159831,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}