{"id":"W4396650996","doi":"10.1145/3640457.3688142","title":"Bayesian Optimization with LLM-Based Acquisition Functions for Natural Language Preference Elicitation","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; University of Waterloo","funders":"","keywords":"Preference elicitation; Preference; Computer science; Bayesian optimization; Natural language; Natural (archaeology); Artificial intelligence; Natural language processing; Mathematics; Statistics; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00600427,0.001582084,0.001694198,0.001165949,0.0007120683,0.001683667,0.002385386,0.0019852,0.006728468],"category_scores_gemma":[0.03198728,0.001062988,0.001247301,0.001140213,0.001759805,0.003680911,0.003441146,0.003746912,0.001548777],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001955719,"about_ca_system_score_gemma":0.002813433,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005013525,"about_ca_topic_score_gemma":0.007720393,"domain_scores_codex":[0.9953868,0.00281927,0.0002056665,0.0005823818,0.0007390609,0.0002669316],"domain_scores_gemma":[0.9835112,0.01385215,0.0005963223,0.0007322693,0.0009755963,0.0003324962],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004310569,0.0003454377,0.002842537,0.0003982205,0.0001261845,0.0001952183,0.0007335105,0.6784489,0.004185351,0.1017843,0.005096183,0.2054131],"study_design_scores_gemma":[0.00002444257,0.0000479433,0.0001287484,0.00002158651,0.000008766718,0.00002148595,0.00002894285,0.9690506,0.0007029097,0.02921377,0.0007363037,0.00001447487],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005004246,0.00009230257,0.9928306,0.0002487753,0.000009357635,0.00009504177,0.00005486769,0.0003592049,0.001305595],"genre_scores_gemma":[0.2920647,0.0001850563,0.7013893,0.0007316184,0.00006165817,0.001098204,0.0004419184,0.0003314879,0.003696098],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006728468,"threshold_uncertainty_score":0.03175402,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01651333121361138,"score_gpt":0.2480529212065353,"score_spread":0.231539589992924,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}