{"id":"W6976738005","doi":"10.60692/10q1k-4bd19","title":"Can a Gorilla Ride a Camel? Learning Semantic Plausibility from Text","year":2019,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Usability and User Interface Design","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; McGill University","funders":"","keywords":"Natural language understanding; Language model; Set (abstract data type); Commonsense knowledge; Language understanding; Commonsense reasoning; Natural language; Testbed","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002762861,0.0007702697,0.0009882697,0.0008200274,0.001181926,0.002434588,0.001644714,0.001615853,0.01529512],"category_scores_gemma":[0.01629817,0.0005147959,0.001019782,0.000606976,0.001695945,0.006876776,0.002145654,0.002660177,0.003586411],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006597451,"about_ca_system_score_gemma":0.0007814979,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002488038,"about_ca_topic_score_gemma":0.004000474,"domain_scores_codex":[0.999006,0.0003598464,0.00006299532,0.0003514918,0.0001567104,0.00006304305],"domain_scores_gemma":[0.9982336,0.001110495,0.00005906491,0.0003874027,0.0001389749,0.00007044808],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00204199,0.0002774738,0.00886067,0.001387616,0.0005107412,0.001737821,0.002397159,0.04479012,0.01299006,0.2019542,0.1151692,0.6078829],"study_design_scores_gemma":[0.0002820152,0.0002899502,0.002288283,0.0002964594,0.0001869151,0.0009512791,0.001005443,0.3447537,0.01315517,0.5162786,0.1203573,0.0001548612],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.159585,0.002394708,0.7362095,0.02477624,0.001633019,0.0002055185,0.004541307,0.02638177,0.04427303],"genre_scores_gemma":[0.692414,0.0005383889,0.2837914,0.003975364,0.000243518,0.0001154596,0.005007491,0.002185136,0.01172932],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01529512,"threshold_uncertainty_score":0.05116725,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02524745682790789,"score_gpt":0.202151425145915,"score_spread":0.1769039683180071,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}