{"id":"W2054795804","doi":"10.1109/icmla.2012.31","title":"An Inverse Reinforcement Learning Algorithm for Partially Observable Domains with Application on Healthcare Dialogue Management","year":2012,"lang":"en","type":"article","venue":"","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"McGill University","keywords":"Partially observable Markov decision process; Markov decision process; Computer science; Reinforcement learning; Margin (machine learning); Artificial intelligence; Observable; Machine learning; Inverse; Markov process; Markov chain; Mathematical optimization; Markov model; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005062032,0.0001467488,0.0001418769,0.00007350749,0.0002227785,0.00009380627,0.0003745792,0.00005118121,0.000004206556],"category_scores_gemma":[0.000004210663,0.0001160011,0.00003701572,0.0002074005,0.00001583892,0.0005728668,0.00006089863,0.00007157258,0.00009438489],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001043071,"about_ca_system_score_gemma":0.00003462849,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000369043,"about_ca_topic_score_gemma":0.0002040315,"domain_scores_codex":[0.9985983,0.00006331052,0.0002120703,0.0003319224,0.0002935013,0.0005008842],"domain_scores_gemma":[0.9989676,0.00003625406,0.0001068124,0.0005534448,0.0000668905,0.0002690321],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001191663,0.0003936501,0.00653003,0.00017714,0.00009599926,0.000006827324,0.002210418,0.01684408,0.0002487452,0.5698799,0.003048059,0.4004459],"study_design_scores_gemma":[0.003034524,0.00307692,0.004418836,0.00006933939,0.000031482,0.00001066512,0.0009391754,0.8685197,0.003893164,0.001371402,0.1138378,0.0007969021],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001292724,0.00001447502,0.9932789,0.0003307272,0.0002493691,0.001106996,0.000002011642,0.0002382731,0.003486509],"genre_scores_gemma":[0.7417098,0.0000084911,0.2554078,0.001115888,0.0002916803,0.0006407892,0.0001213572,0.00001588858,0.0006883605],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8516757,"threshold_uncertainty_score":0.4730389,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02451411383205607,"score_gpt":0.2634534985663381,"score_spread":0.238939384734282,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}