{"id":"W4385574362","doi":"10.18653/v1/2022.blackboxnlp-1.27","title":"Using Roark-Hollingshead Distance to Probe BERT’s Syntactic Competence","year":2022,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Syntax; Natural language processing; Artificial intelligence; Language model; Encoder; Conjecture; Syntactic structure; Baseline (sea); Linguistics; Philosophy; Mathematics; Discrete mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001901577,0.0007639825,0.0005670876,0.001040035,0.0006942684,0.001962381,0.001024532,0.00130119,0.006243001],"category_scores_gemma":[0.01701177,0.0003503464,0.0005023564,0.0006832634,0.001596189,0.004602258,0.002427093,0.002204075,0.001728565],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008073804,"about_ca_system_score_gemma":0.0007600316,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002689744,"about_ca_topic_score_gemma":0.004371311,"domain_scores_codex":[0.9990206,0.0002864471,0.00005642468,0.0003494787,0.0001922572,0.000094817],"domain_scores_gemma":[0.9930864,0.004440885,0.0005152975,0.001182021,0.0004743319,0.0003010227],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001620013,0.0002922166,0.06824749,0.0006170135,0.0002006633,0.0007330914,0.006657172,0.08270904,0.08085704,0.2473807,0.01284706,0.4978385],"study_design_scores_gemma":[0.00006552483,0.0004326316,0.038382,0.00009159099,0.00007893656,0.0007407746,0.001615048,0.5488057,0.0262336,0.3712003,0.01216126,0.0001927578],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5150545,0.0003186002,0.454948,0.001019723,0.00006448841,0.00006664854,0.001097185,0.002026047,0.02540478],"genre_scores_gemma":[0.9318601,0.00006793864,0.06389815,0.0001999284,0.00001484016,0.00005648809,0.001099269,0.0003665749,0.00243665],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.006243001,"threshold_uncertainty_score":0.02088487,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02898207653309455,"score_gpt":0.294284269398719,"score_spread":0.2653021928656245,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}