{"id":"W7086984966","doi":"10.1162/tacl.a.38","title":"Elements of World Knowledge (<scp>EWoK</scp>): A Cognition-Inspired Framework for Evaluating Basic World Knowledge in Language Models","year":2025,"lang":"en","type":"article","venue":"Transactions of the Association for Computational Linguistics","topic":"Schizophrenia research and treatment","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute","funders":"Quest for Intelligence, Massachusetts Institute of Technology; Yuhan","keywords":"Situated; Language model; Conceptual model; Domain knowledge; Simple (philosophy); Conceptual framework; Knowledge-based systems","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006809917,0.0001404136,0.0003381455,0.0005329085,0.0001809124,0.00001567271,0.0001311821,0.00008196464,0.00000817421],"category_scores_gemma":[0.004389526,0.0001296965,0.0002806904,0.001030152,0.00004064945,0.000027641,0.00001302607,0.0002058895,0.000002552859],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000481377,"about_ca_system_score_gemma":0.0007345375,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002011318,"about_ca_topic_score_gemma":0.0006790148,"domain_scores_codex":[0.9984742,0.00009306655,0.0006381486,0.0002025556,0.0003564021,0.0002356349],"domain_scores_gemma":[0.9923193,0.005277524,0.0003402986,0.0001639657,0.001845659,0.00005328325],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003656105,0.009133409,0.03150199,0.003483314,0.004862418,0.000001833367,0.005508846,0.2483558,0.000376326,0.6652368,0.003272965,0.02461015],"study_design_scores_gemma":[0.01258457,0.0003826045,0.02892273,0.001531951,0.001391677,4.707865e-7,0.0005070637,0.5304009,0.003561498,0.4196565,0.0009312079,0.0001289029],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.104884,0.001567821,0.8662854,0.0009351855,0.001926601,0.006354967,0.002615128,0.0001197278,0.01531123],"genre_scores_gemma":[0.9097208,0.000005202445,0.08556057,0.00003919172,0.0001087052,0.0002719188,0.0002089466,0.00001873222,0.004065898],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8048369,"threshold_uncertainty_score":0.5288868,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04625059357637484,"score_gpt":0.3867130782903953,"score_spread":0.3404624847140204,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}