{"id":"W4385374194","doi":"10.48550/arxiv.2307.14993","title":"Thinker: Learning to Plan and Act","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council; Science and Technology Facilities Council; Université de Montréal; Dell EMC","keywords":"Plan (archaeology); Computer science; Benchmark (surveying); Reinforcement learning; Artificial intelligence; Action (physics); Interpretation (philosophy); State (computer science); Machine learning; Automated planning and scheduling; Algorithm; Programming language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0002970322,0.0002354068,0.0002430393,0.0003001754,0.0001976031,0.0002295329,0.001568033,0.0002075903,0.00001604524],"category_scores_gemma":[0.000136846,0.00027645,0.00008918856,0.0005147904,0.00007837017,0.0002609887,0.003845101,0.0006651685,0.0007412804],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009809236,"about_ca_system_score_gemma":0.00009892724,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003421689,"about_ca_topic_score_gemma":0.0001351026,"domain_scores_codex":[0.998237,0.0001098105,0.000155582,0.001062619,0.0000932844,0.000341767],"domain_scores_gemma":[0.9986007,0.000222628,0.0001181883,0.000761555,0.00008217668,0.0002147709],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003222155,0.00004419594,0.02968617,0.00007210447,0.0001186788,0.0009313985,0.006340608,0.7625729,0.00009390894,0.1885878,0.001518571,0.0100014],"study_design_scores_gemma":[0.00008326829,0.0001296478,0.005740905,0.0001786702,0.000039379,0.00000665505,0.0007685726,0.865802,0.0006405044,0.1216645,0.004153064,0.0007927876],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.458968,0.00003214544,0.5366115,0.0005261525,0.0008266841,0.0002604629,0.00000429516,0.0008591577,0.00191165],"genre_scores_gemma":[0.9922762,0.00009604031,0.001932665,0.0001206408,0.00007433477,7.976428e-7,0.000003722334,0.00002003379,0.005475563],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5346788,"threshold_uncertainty_score":0.9999688,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1591584466283779,"score_gpt":0.2243870914711452,"score_spread":0.06522864484276725,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}