{"id":"W7086984966","doi":"10.1162/tacl.a.38","title":"Elements of World Knowledge (<scp>EWoK</scp>): A Cognition-Inspired Framework for Evaluating Basic World Knowledge in Language Models","year":2025,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Schizophrenia research and treatment","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute","funders":"Quest for Intelligence, Massachusetts Institute of Technology; Yuhan","keywords":"Situated; Language model; Conceptual model; Domain knowledge; Simple (philosophy); Conceptual framework; Knowledge-based systems","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01025439,0.00185831,0.0009416936,0.008193426,0.001080158,0.005221235,0.002839415,0.002955503,0.005328209],"category_scores_gemma":[0.05874512,0.0005915812,0.002021406,0.004879477,0.001860841,0.006650415,0.005277615,0.003134143,0.001584124],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002005639,"about_ca_system_score_gemma":0.001672002,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01032883,"about_ca_topic_score_gemma":0.01636918,"domain_scores_codex":[0.9910358,0.004647004,0.001010058,0.001404489,0.001651868,0.0002507439],"domain_scores_gemma":[0.9636072,0.02502816,0.002160604,0.00600261,0.002425677,0.0007757291],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001653832,0.001669878,0.06574433,0.003428713,0.001601136,0.0007277629,0.001853402,0.2641517,0.005503545,0.06933048,0.1005255,0.4838098],"study_design_scores_gemma":[0.0001458486,0.0003421513,0.01136979,0.0003997107,0.000137939,0.0002646047,0.0005839518,0.8748009,0.004959216,0.0806859,0.02615811,0.0001518009],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1754205,0.004024954,0.6953224,0.003018302,0.0005002106,0.001885175,0.08051805,0.01807152,0.02123893],"genre_scores_gemma":[0.5527543,0.0006855269,0.3659405,0.0005863734,0.0001261809,0.001309829,0.07644338,0.0007126877,0.001441185],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01032883,"threshold_uncertainty_score":0.05423111,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04625059357637484,"score_gpt":0.3867130782903953,"score_spread":0.3404624847140204,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}