{"id":"W4414973250","doi":"10.48550/arxiv.2510.05026","title":"Idiom Understanding as a Tool to Measure the Dialect Gap","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Linguistics and Discourse Analysis","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Benchmark (surveying); Set (abstract data type); Metropolitan area; Construct (python library); Data set; Natural language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003966757,0.00158474,0.0007695811,0.00747662,0.001221273,0.003235739,0.001369759,0.001453393,0.004527248],"category_scores_gemma":[0.01760461,0.0002718937,0.00077857,0.003498729,0.0009371994,0.004870104,0.003482301,0.00193715,0.002164339],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001466622,"about_ca_system_score_gemma":0.001231271,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02506709,"about_ca_topic_score_gemma":0.03173851,"domain_scores_codex":[0.9964977,0.00135214,0.0003333219,0.0009571774,0.0005744924,0.0002851567],"domain_scores_gemma":[0.9893964,0.005787436,0.0006300835,0.001856653,0.001811425,0.0005180764],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001857632,0.0008532378,0.3870499,0.00233377,0.001036336,0.001162545,0.01673081,0.03527281,0.03254171,0.02114473,0.04842157,0.4515949],"study_design_scores_gemma":[0.0002655463,0.001002589,0.3935072,0.0006542248,0.000600449,0.003000387,0.02496491,0.3616755,0.04473861,0.04265592,0.1265779,0.0003566531],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8384829,0.003343748,0.08376154,0.001000379,0.0001297819,0.0003744172,0.02678928,0.004867129,0.04125093],"genre_scores_gemma":[0.9326928,0.0002949605,0.03411366,0.0002007976,0.00003359823,0.0002108629,0.03016977,0.0003074056,0.001976176],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02506709,"threshold_uncertainty_score":0.04984236,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2309519211307921,"score_gpt":0.3185835870538066,"score_spread":0.0876316659230145,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}