{"id":"W6910551761","doi":"10.48448/86bh-9021","title":"Can Language Models Serve as Text-Based World Simulators?","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Merck Canada Inc. (Canada)","funders":"","keywords":"Benchmarking; Context (archaeology); Key (lock); Benchmark (surveying); Virtual world; Work (physics); Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004557669,0.002472729,0.001083669,0.001918793,0.000628118,0.003397668,0.003927861,0.002615473,0.006562707],"category_scores_gemma":[0.03492451,0.0007435487,0.001260003,0.001648881,0.001088207,0.007324675,0.002628036,0.00281376,0.005923716],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001763994,"about_ca_system_score_gemma":0.002016901,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01543018,"about_ca_topic_score_gemma":0.02068597,"domain_scores_codex":[0.9956097,0.002091793,0.0003453811,0.0009845397,0.0007039219,0.0002647197],"domain_scores_gemma":[0.9847,0.00948353,0.0007086532,0.00331422,0.001200787,0.0005928914],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001913315,0.00105134,0.03168581,0.002100971,0.0005814998,0.0005438117,0.0006365681,0.5382656,0.004427137,0.01918469,0.1429355,0.2566737],"study_design_scores_gemma":[0.0001580738,0.0002835557,0.00194576,0.0001513652,0.00006634132,0.0001265869,0.0002312424,0.9539533,0.003601581,0.02004697,0.01937143,0.00006374832],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4132701,0.007753113,0.3067398,0.01166939,0.00296797,0.001113765,0.1325987,0.08237236,0.04151484],"genre_scores_gemma":[0.7233295,0.001235777,0.1260849,0.001664215,0.0001562855,0.0005432384,0.1394351,0.002183993,0.005367053],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01543018,"threshold_uncertainty_score":0.03068072,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02412718222446021,"score_gpt":0.3182231397096635,"score_spread":0.2940959574852033,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}