{"id":"W4416035291","doi":"10.18653/v1/2025.emnlp-main.1712","title":"Mechanisms vs. Outcomes: Probing for Syntax Fails to Explain Performance on Targeted Syntactic Evaluations","year":2025,"lang":"","type":"article","venue":"","topic":"Neurobiology of Language and Bilingualism","field":"Neuroscience","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Canadian Institute for Advanced Research","keywords":"Syntax; Semantics (computer science); Control (management); Feature (linguistics); Class (philosophy)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008056136,0.001735335,0.001118268,0.0009232246,0.0004364165,0.003264363,0.001688013,0.002453789,0.005620783],"category_scores_gemma":[0.03194995,0.0007440874,0.001066053,0.0004428093,0.002160053,0.007424679,0.001953481,0.003006266,0.001318947],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005995136,"about_ca_system_score_gemma":0.001559788,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001688616,"about_ca_topic_score_gemma":0.001830727,"domain_scores_codex":[0.9967998,0.0007839815,0.0002891126,0.0008957407,0.0007712017,0.0004601609],"domain_scores_gemma":[0.9766265,0.01078024,0.005114948,0.005549566,0.0009081086,0.001020728],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003246722,0.002028115,0.6350243,0.0008263925,0.001459875,0.001586422,0.007159115,0.008634748,0.1466706,0.0518821,0.002716097,0.1387655],"study_design_scores_gemma":[0.000214613,0.001381096,0.7545693,0.0001618871,0.0005630565,0.001437912,0.001955468,0.03645257,0.02043867,0.1809761,0.001656603,0.0001927238],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.96952,0.0001688143,0.01582978,0.001066432,0.00006051816,0.00006929863,0.0004039223,0.00025441,0.01262677],"genre_scores_gemma":[0.9970957,0.00004309725,0.001687319,0.00009816061,0.00001260237,0.00003253107,0.0001671994,0.0001557758,0.0007075485],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008056136,"threshold_uncertainty_score":0.0426054,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03884629213074269,"score_gpt":0.3507860355476899,"score_spread":0.3119397434169472,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}