{"id":"W4389363042","doi":"10.48550/arxiv.2312.00949","title":"Hyperparameter Optimization for Large Language Model Instruction-Tuning","year":2023,"lang":"en","type":"preprint","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs; Hydro-Québec","keywords":"Hyperparameter; Computer science; Pipeline (software); Rank (graph theory); Set (abstract data type); Decomposition; Adaptation (eye); Machine learning; Language model; Artificial intelligence; Fine-tuning; Adaptive optimization; Programming language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003193746,0.001919726,0.001312456,0.0009628356,0.0007686327,0.001758243,0.002134648,0.002175207,0.006524996],"category_scores_gemma":[0.01973183,0.0009708243,0.001033875,0.0008061369,0.001225421,0.00251136,0.002415411,0.004455136,0.003238484],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001385888,"about_ca_system_score_gemma":0.001617622,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0042709,"about_ca_topic_score_gemma":0.008835417,"domain_scores_codex":[0.9982241,0.0009782917,0.00007932029,0.0003469614,0.0002079315,0.0001633867],"domain_scores_gemma":[0.9950904,0.003549261,0.0001549529,0.0007374688,0.0003419796,0.0001259203],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003097058,0.0002282707,0.001659657,0.0002800932,0.0001907371,0.0001787646,0.0003155288,0.7772195,0.007381352,0.0159146,0.01255711,0.1837647],"study_design_scores_gemma":[0.00003143822,0.00001932474,0.0001216846,0.0000211402,0.00001239086,0.0000203888,0.00002966943,0.9808836,0.001765719,0.01560105,0.001481838,0.00001174718],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03191505,0.001357257,0.9512773,0.000710111,0.000157215,0.0001508767,0.0002815416,0.009854615,0.004295959],"genre_scores_gemma":[0.5460009,0.0005069893,0.4408585,0.001235028,0.0001445406,0.0008452594,0.001286925,0.003820097,0.005301707],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006524996,"threshold_uncertainty_score":0.02182835,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02511101953598357,"score_gpt":0.2599844484148164,"score_spread":0.2348734288788328,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}