{"id":"W6947992274","doi":"10.48448/qj49-ww69","title":"A fine-grained comparison of pragmatic language understanding in humans and language models","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Species Distribution and Climate Change","field":"Environmental Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Pragmatics; Set (abstract data type); Literal (mathematical logic); Language model; Human language; Computational model; Natural language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004706147,0.0004968356,0.0004076232,0.0007139746,0.0004619294,0.002892512,0.0006887722,0.001115305,0.004829238],"category_scores_gemma":[0.03056153,0.0003971792,0.0003777798,0.0002970463,0.001900036,0.004603147,0.002512817,0.001029889,0.001132181],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006802123,"about_ca_system_score_gemma":0.0006088516,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002480502,"about_ca_topic_score_gemma":0.002856102,"domain_scores_codex":[0.9958917,0.002280653,0.0001917476,0.0009205995,0.0005780151,0.0001372957],"domain_scores_gemma":[0.981298,0.01347744,0.0008321088,0.003191187,0.0008757329,0.0003256423],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002247621,0.001745897,0.1672627,0.001898112,0.0004796416,0.0009753911,0.08705985,0.05763029,0.2386036,0.06294558,0.020832,0.3583194],"study_design_scores_gemma":[0.0002564987,0.002290309,0.2708919,0.0004237785,0.0001827254,0.001719603,0.03389149,0.4202444,0.08626574,0.1329803,0.0502189,0.0006344354],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8885122,0.0002023921,0.0863061,0.0006959378,0.0000555547,0.000199717,0.001226683,0.0015401,0.02126128],"genre_scores_gemma":[0.9721468,0.00006379015,0.02537286,0.0001607963,0.000008082705,0.0001085741,0.0008895458,0.0001700105,0.001079557],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004829238,"threshold_uncertainty_score":0.02488875,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05042189979535362,"score_gpt":0.3155308244965832,"score_spread":0.2651089247012296,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}