{"id":"W4391013074","doi":"10.48550/arxiv.2401.08898","title":"Bridging State and History Representations: Understanding Self-Predictive RL","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; McGill University; DeepMind; Nvidia","keywords":"Bridging (networking); State (computer science); Computer science; Algorithm; Computer security","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0002566829,0.0002422062,0.0002293785,0.0003905549,0.0001249603,0.0001512457,0.0007401512,0.0001355875,0.00001251639],"category_scores_gemma":[0.00002173867,0.0003052591,0.0001040454,0.0002555872,0.00008754139,0.0003626662,0.003091217,0.0006635318,0.00003581797],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001973483,"about_ca_system_score_gemma":0.0003345454,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002448782,"about_ca_topic_score_gemma":0.00002808831,"domain_scores_codex":[0.9980135,0.0001054965,0.0001903272,0.001301623,0.0001090399,0.0002800218],"domain_scores_gemma":[0.9986787,0.0001007164,0.0001469291,0.000864798,0.00006621525,0.0001426392],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001603405,0.00004334727,0.001540356,0.0003626086,0.0003343287,0.001078613,0.007411666,0.292452,0.00002317489,0.6936331,0.002115683,0.0009890604],"study_design_scores_gemma":[0.0001445892,0.0000122583,0.0001105909,0.000125657,0.00005978083,0.000007079577,0.000214912,0.7753838,0.00000815562,0.2234865,0.0002152032,0.0002314911],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05230718,0.0007064221,0.9374859,0.0001772673,0.001230241,0.0002254086,0.00000987848,0.0005240653,0.00733364],"genre_scores_gemma":[0.9932407,0.000400045,0.003826329,0.00006437462,0.00007616715,0.00000113075,0.000003296525,0.00001922356,0.002368751],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9409335,"threshold_uncertainty_score":0.99994,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1209434695660798,"score_gpt":0.201324250554225,"score_spread":0.08038078098814515,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}