{"id":"W6929418907","doi":"10.48448/g3qz-q748","title":"Multilingual Nonce Dependency Treebanks: Understanding how Language Models Represent and Process Syntactic Structure","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Freshwater macroinvertebrate diversity and ecology","field":"Environmental Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Cryptographic nonce; Perplexity; Grammaticality; Syntax; Dependency (UML); Word (group theory); Argument (complex analysis); Lexicalization","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0003188215,0.0003479231,0.0002989506,0.0003047417,0.0002723895,0.000308697,0.0007181771,0.0002962242,0.002835886],"category_scores_gemma":[0.00006543155,0.0003029408,0.00004249178,0.0005473649,0.001563332,0.0005977093,0.0007569852,0.0004966739,0.0002561242],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004770098,"about_ca_system_score_gemma":0.00009745386,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00130031,"about_ca_topic_score_gemma":0.01499838,"domain_scores_codex":[0.9973006,0.00004379652,0.0001653075,0.001178581,0.0006902809,0.0006214104],"domain_scores_gemma":[0.9991779,0.00003486074,0.0001407144,0.0004141448,0.00001059155,0.0002217401],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002727067,0.001114003,0.02337468,0.005003445,0.0009480463,0.007010505,0.1043159,0.03078931,0.05310048,0.01510287,0.7221598,0.03680822],"study_design_scores_gemma":[0.003185318,0.000892549,0.0003011955,0.001519511,0.0008725193,0.000998351,0.07747691,0.7504467,0.01426131,0.1276805,0.01648806,0.005877077],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.2047572,0.004214551,0.00845996,0.004137448,0.00495306,0.003073744,0.001047187,0.001865037,0.7674918],"genre_scores_gemma":[0.9033662,0.0000499113,0.001091952,0.000196255,0.00010749,0.000004230146,0.00002204726,0.0001251726,0.09503673],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7196574,"threshold_uncertainty_score":0.9999422,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02489636936238457,"score_gpt":0.2618803352903604,"score_spread":0.2369839659279758,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}