{"id":"W6891723697","doi":"10.48448/rqwd-cy29","title":"Few-shot Fine-tuning vs. In-context Learning: A Fair Comparison and Evaluation","year":2023,"lang":"en","type":"other","venue":"Open MIND","topic":"Domain Adaptation and Few-Shot Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Set (abstract data type); Identification (biology); Process (computing); Quality (philosophy); Term (time)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0132241,0.00511032,0.003609748,0.002789409,0.001299147,0.00311177,0.005188733,0.005380774,0.005035411],"category_scores_gemma":[0.03398857,0.0009065256,0.001979863,0.001560881,0.001965127,0.006644247,0.003878659,0.004071628,0.004126709],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002238095,"about_ca_system_score_gemma":0.001809956,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01270183,"about_ca_topic_score_gemma":0.01455229,"domain_scores_codex":[0.9910775,0.003323016,0.0006138145,0.002565454,0.0017294,0.0006907913],"domain_scores_gemma":[0.9856867,0.007216204,0.0005441705,0.003833243,0.001813072,0.0009066249],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.008145208,0.003415348,0.009672969,0.004883213,0.002272206,0.0005415812,0.0004346678,0.2656344,0.01674694,0.00269793,0.04938187,0.6361737],"study_design_scores_gemma":[0.001266289,0.005720505,0.01078622,0.0007252506,0.0007831568,0.001388089,0.0006086869,0.9268914,0.02356783,0.008080246,0.01982824,0.0003542202],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4927589,0.0973678,0.2830476,0.003478925,0.005464272,0.004481807,0.01327574,0.06494079,0.0351841],"genre_scores_gemma":[0.7468045,0.006378099,0.2006075,0.002306236,0.0007678393,0.001504831,0.02747191,0.003375109,0.01078394],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0132241,"threshold_uncertainty_score":0.06993651,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1223040697787293,"score_gpt":0.3728091254878406,"score_spread":0.2505050557091113,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}