{"id":"W4385573416","doi":"10.18653/v1/2022.findings-emnlp.393","title":"Detecting Languages Unintelligible to Multilingual Models through Local Structure Probes","year":2022,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; Huawei Technologies (Canada); Polytechnique Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language processing; Artificial intelligence; Set (abstract data type); Variety (cybernetics); Task (project management); Construct (python library); Sentence; Transfer (computing); Transfer of learning; Language model; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002368634,0.0012151,0.0007232144,0.001096724,0.001253461,0.002877219,0.001345975,0.001252138,0.00280714],"category_scores_gemma":[0.008933166,0.0005598238,0.001285563,0.0009217391,0.001103394,0.004221437,0.002983579,0.002814733,0.001879761],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001182613,"about_ca_system_score_gemma":0.00131344,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004417529,"about_ca_topic_score_gemma":0.01153191,"domain_scores_codex":[0.9983438,0.0006553766,0.00006455449,0.0006523492,0.0001368762,0.0001470935],"domain_scores_gemma":[0.9953366,0.002732203,0.000334003,0.0008801711,0.000499527,0.0002175313],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002008412,0.000825842,0.0696718,0.0009846617,0.001071747,0.002535437,0.008699149,0.1348149,0.1389005,0.04089993,0.03117905,0.5684085],"study_design_scores_gemma":[0.00007271177,0.0003120964,0.009518404,0.0001154256,0.0002861645,0.0006569007,0.003274491,0.8745674,0.03584353,0.06012968,0.01508885,0.0001343432],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5245926,0.001002618,0.4522071,0.001579432,0.0001727476,0.0001675069,0.001555388,0.007387058,0.01133553],"genre_scores_gemma":[0.9360279,0.0001890943,0.05682034,0.0005104651,0.00005994657,0.000118169,0.002765673,0.0007029658,0.002805508],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004417529,"threshold_uncertainty_score":0.01252663,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02989026448838823,"score_gpt":0.2909275560061457,"score_spread":0.2610372915177575,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}