{"id":"W4389524537","doi":"10.18653/v1/2023.findings-emnlp.822","title":"Efficiently Enhancing Zero-Shot Performance of Instruction Following Model via Retrieval of Soft Prompt","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; Korea Advanced Institute of Science and Technology","keywords":"Computer science; Embedding; Benchmark (surveying); Generalization; Task (project management); Inference; Zero (linguistics); Shot (pellet); Artificial intelligence; Scaling; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005124356,0.00009898526,0.0001922907,0.000192321,0.00006982493,0.00001845307,0.0004613681,0.00005669575,0.00000295596],"category_scores_gemma":[0.00004193619,0.00009328175,0.00008760596,0.0007654746,0.00002115279,0.0004214483,0.0002427596,0.00009252918,0.000009658866],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003242003,"about_ca_system_score_gemma":0.00008700147,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001800338,"about_ca_topic_score_gemma":0.000002152546,"domain_scores_codex":[0.9986001,0.00001704949,0.0004108851,0.0002967183,0.0004379254,0.0002373698],"domain_scores_gemma":[0.99928,0.00003705254,0.0001257292,0.0004315964,0.00008469136,0.00004090791],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002131971,0.00003767282,0.001442742,0.0001840158,0.00002394484,0.000001927981,0.001788396,0.6898988,0.2764483,0.005761885,0.00001535537,0.02437567],"study_design_scores_gemma":[0.0001948604,0.00004613335,0.0002502703,0.00004588856,0.000003944772,0.000002082022,0.00001874891,0.8152128,0.1835392,0.0006080839,0.000001871609,0.00007609391],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5106805,0.000005753696,0.4886642,0.00001729966,0.0002118859,0.00006883717,2.043024e-7,0.00009636062,0.0002549984],"genre_scores_gemma":[0.9440622,0.000002169946,0.05566628,0.00001469885,0.00001646484,0.000002112491,9.59397e-7,0.000007284275,0.0002277793],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4333818,"threshold_uncertainty_score":0.3803919,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02492057637310018,"score_gpt":0.2507083064115961,"score_spread":0.2257877300384959,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}