{"id":"W7132894057","doi":"","title":"Discount Factor Estimation in Inverse Reinforcement Learning","year":2022,"lang":"","type":"dissertation","venue":"TSpace","topic":"Traffic control and management","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Discounting; Reinforcement learning; Factor (programming language); Process (computing); Computation; Entropy (arrow of time); Estimation","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00394937,0.001295717,0.001673304,0.0006194306,0.0003456436,0.001501349,0.001593691,0.001413858,0.002157319],"category_scores_gemma":[0.01424768,0.0006769881,0.0007686166,0.0004589978,0.002006297,0.002090688,0.001719983,0.002792131,0.0003895563],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001646697,"about_ca_system_score_gemma":0.001565322,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004063751,"about_ca_topic_score_gemma":0.002332713,"domain_scores_codex":[0.9986455,0.0006252257,0.00007381653,0.0002693716,0.0002673987,0.0001187528],"domain_scores_gemma":[0.9946196,0.004219021,0.0003480614,0.0001840802,0.0004819774,0.000147172],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008007861,0.00004822793,0.0008443835,0.0001154164,0.00004106054,0.0000622699,0.0001079112,0.9227865,0.0006549825,0.04349927,0.0004850329,0.031275],"study_design_scores_gemma":[0.000007931398,0.00002183366,0.00004239006,0.00001035974,0.000005059072,0.000007947586,0.000004126646,0.9854932,0.0002116215,0.01397575,0.0002138003,0.000005930651],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006925832,0.0002781858,0.9911384,0.0001683528,0.00002957422,0.00003624512,0.00001532196,0.0001225047,0.001285572],"genre_scores_gemma":[0.7291521,0.0005467556,0.2651552,0.0002688007,0.0000809248,0.0002640751,0.0001064522,0.0001541041,0.004271526],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004063751,"threshold_uncertainty_score":0.02088654,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01045129542576565,"score_gpt":0.2744178015171125,"score_spread":0.2639665060913469,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}