{"id":"W7039045101","doi":"","title":"Learning state representations in reinforcement learning using mixed policy successor features","year":2021,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Historical and Architectural Studies","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Prior probability; Successor cardinal; Sample (material); Inefficiency; Initialization; Feature (linguistics); ENCODE","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001447762,0.0006265458,0.0008506561,0.0004657706,0.0003105963,0.0008521264,0.00108359,0.0009310277,0.00192455],"category_scores_gemma":[0.006418787,0.0005249627,0.000431301,0.0004024958,0.0009886594,0.001785975,0.000970951,0.001754506,0.0003085386],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001054941,"about_ca_system_score_gemma":0.001002242,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003530298,"about_ca_topic_score_gemma":0.00468322,"domain_scores_codex":[0.9994898,0.0002190544,0.00002382153,0.0001253129,0.00008274436,0.00005928345],"domain_scores_gemma":[0.9977581,0.001613116,0.0001954272,0.0001693205,0.0001682594,0.00009572165],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001381331,0.00009238905,0.001532354,0.00004261835,0.00002672319,0.00005210327,0.00007330388,0.9256933,0.001143482,0.01020942,0.0007284384,0.06026762],"study_design_scores_gemma":[0.000009387566,0.00002273927,0.00008487947,0.000005033257,0.000002958386,0.000005794802,0.000003239671,0.994718,0.0002798039,0.004746898,0.0001179041,0.000003441256],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.08266698,0.0002255196,0.9140494,0.000336527,0.00003046443,0.00006212716,0.0001013129,0.0008507587,0.001676863],"genre_scores_gemma":[0.886029,0.00009819648,0.111464,0.0001647268,0.00002260344,0.0001573215,0.0001903291,0.00007629026,0.001797515],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003530298,"threshold_uncertainty_score":0.007656634,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02790454777727037,"score_gpt":0.2665603426383485,"score_spread":0.2386557948610782,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}