{"id":"W2900994167","doi":"","title":"A Fitted-Q Algorithm for Budgeted MDPs","year":2018,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Algorithm; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.005275964,0.0004640111,0.0004695448,0.0002890685,0.0005081773,0.001139643,0.004370029,0.0004305861,0.00006885665],"category_scores_gemma":[0.001761834,0.0005090042,0.0003288083,0.0005064955,0.0002617679,0.0002847003,0.003865024,0.0007074056,0.0001286231],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001796627,"about_ca_system_score_gemma":0.0004649127,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002225988,"about_ca_topic_score_gemma":0.00006818111,"domain_scores_codex":[0.9943461,0.002451952,0.0007117121,0.001220323,0.0006466612,0.0006232712],"domain_scores_gemma":[0.9894705,0.001744118,0.0007880071,0.004040226,0.003717708,0.0002394568],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001835839,0.0009342554,0.0006361443,0.0007286614,0.0006298505,0.00001854719,0.02272661,0.01787254,0.001044333,0.2423069,0.04209709,0.6709867],"study_design_scores_gemma":[0.0005226999,0.000001222553,0.0003172823,0.0008230967,0.00003272129,0.000007529537,0.00001708698,0.944098,0.00978004,0.005127728,0.03872993,0.0005426607],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0006623357,0.0002570533,0.9737176,0.005584324,0.0008454506,0.0008723104,0.00003766498,0.0007026271,0.01732058],"genre_scores_gemma":[0.02794163,0.0001728356,0.951548,0.0003116884,0.0001043891,0.0002256491,0.0004977182,0.00007528849,0.01912284],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9262255,"threshold_uncertainty_score":0.9998972,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0166242353117175,"score_gpt":0.2425659507311312,"score_spread":0.2259417154194137,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}