{"id":"W6957621777","doi":"10.60692/5r3jg-m5e18","title":"Imperfect also Deserves Reward: Multi-Level and Sequential Reward Modeling for Better Dialog Management","year":2021,"lang":"en","type":"article","venue":"Greater South Information System","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Imperfect; Dialog box; Selection (genetic algorithm); Language model; Association (psychology); Subject (documents)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005143918,0.0001866595,0.0002307612,0.0001680059,0.0001967387,0.0005584469,0.0003325346,0.00009300614,0.000002232171],"category_scores_gemma":[0.00001662251,0.0001700198,0.00009378237,0.0001434529,0.00001135217,0.001296013,0.0003672465,0.000073964,0.00004520797],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008873716,"about_ca_system_score_gemma":0.00003476094,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000008913701,"about_ca_topic_score_gemma":7.911043e-7,"domain_scores_codex":[0.9984312,0.00006651792,0.0005951783,0.0003204882,0.0002679867,0.0003186074],"domain_scores_gemma":[0.9989609,0.000009875281,0.0001493598,0.000582845,0.0002058529,0.00009114813],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002640866,0.00004968017,0.1074805,0.01774725,0.001722698,0.000240762,0.5062873,0.1608049,0.0001886157,0.05247466,0.0008925718,0.151847],"study_design_scores_gemma":[0.001278865,0.00001672888,0.001270652,0.0001686873,0.0000284016,0.00005255968,0.001342773,0.9949448,0.0004810173,0.00003880301,0.0001216865,0.0002550146],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1750169,0.00001116296,0.8234799,0.0001471013,0.0004737766,0.0003711211,0.0000299169,0.0001777753,0.0002923164],"genre_scores_gemma":[0.8314263,5.803138e-7,0.1679907,0.0002763596,0.00007733271,0.00009918181,0.00001430472,0.000008544509,0.0001067244],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8341399,"threshold_uncertainty_score":0.6933205,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09650262403361097,"score_gpt":0.2500986736970738,"score_spread":0.1535960496634629,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}