{"id":"W7019241597","doi":"","title":"Exploring the limits of systematicity of natural language understanding models","year":2023,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Natural language; Natural (archaeology); Normative; Identification (biology); Process (computing)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04099776,0.0009575325,0.001584515,0.002234188,0.002022183,0.008543842,0.003312228,0.002444869,0.004129326],"category_scores_gemma":[0.1986931,0.003196835,0.002766864,0.001931031,0.007144521,0.03739716,0.009961298,0.007938242,0.0007828082],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003454456,"about_ca_system_score_gemma":0.005959102,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009662762,"about_ca_topic_score_gemma":0.01101347,"domain_scores_codex":[0.9618566,0.02811571,0.001642402,0.003538511,0.003992265,0.0008545286],"domain_scores_gemma":[0.5667008,0.3916789,0.004422823,0.02895953,0.006904609,0.001333321],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005404978,0.0003705543,0.01152014,0.0007215629,0.0004505568,0.0002099439,0.005779314,0.07176375,0.002834793,0.7208302,0.0053965,0.1795823],"study_design_scores_gemma":[0.00005095837,0.00007632355,0.0005362671,0.0001137339,0.0001204761,0.00008168838,0.0005292274,0.2915904,0.00174057,0.7004806,0.00465286,0.00002694334],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06722373,0.001953875,0.9105588,0.009764699,0.00009210873,0.0001773467,0.0003899175,0.002047404,0.007792217],"genre_scores_gemma":[0.6869251,0.001501004,0.3046066,0.001376054,0.0002098671,0.0004150539,0.001231395,0.001470357,0.002264608],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04099776,"threshold_uncertainty_score":0.2168195,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1129646712359975,"score_gpt":0.2940237606249028,"score_spread":0.1810590893889053,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}