{"id":"W7009646353","doi":"","title":"Explore the In-context Learning Capability of Large Language Models","year":2024,"lang":"en","type":"dissertation","venue":"UWSpace (University of Waterloo)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Blackberry (Canada)","funders":"","keywords":"Benchmark (surveying); Context (archaeology); Embodied cognition; Domain (mathematical analysis); Software deployment; Natural language understanding; Natural language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002445162,0.001224618,0.0006784755,0.0008712399,0.0006303747,0.001845287,0.001920784,0.001448942,0.001887606],"category_scores_gemma":[0.01353196,0.0005261179,0.001030794,0.0006903188,0.0007738523,0.005376976,0.002620059,0.003349714,0.0008370481],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009993493,"about_ca_system_score_gemma":0.001111834,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005684721,"about_ca_topic_score_gemma":0.01084902,"domain_scores_codex":[0.9985591,0.0007373216,0.00005002586,0.0003987543,0.0001769148,0.00007797801],"domain_scores_gemma":[0.9939704,0.00458632,0.0001838384,0.000808852,0.0003001287,0.0001503987],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004123008,0.0004528722,0.008717913,0.0004006904,0.0003180119,0.0002406638,0.0009540678,0.541209,0.009019199,0.02796431,0.008075449,0.4022354],"study_design_scores_gemma":[0.00001300833,0.00007580912,0.0003377873,0.00001660776,0.00001734486,0.00002872945,0.00008939442,0.9716779,0.00205831,0.02384799,0.001825714,0.00001139993],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2107245,0.003551766,0.7670842,0.002974977,0.0001855451,0.0001589426,0.0009865992,0.007610261,0.006723218],"genre_scores_gemma":[0.8118928,0.0006485161,0.181891,0.0007394475,0.0001290317,0.0001213863,0.002030965,0.0003938792,0.002152976],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005684721,"threshold_uncertainty_score":0.01293141,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01707575341724763,"score_gpt":0.2242240756011204,"score_spread":0.2071483221838728,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}