{"id":"W4389519588","doi":"10.18653/v1/2023.emnlp-main.134","title":"MAGNIFICo: Evaluating the In-Context Learning Ability of Large Language Models to Generalize to Novel Interpretations","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; McGill University","keywords":"Context (archaeology); Computer science; Parsing; Natural language processing; Artificial intelligence; Suite; Natural language; Cognitive science; Linguistics; Cognitive psychology; Psychology; History","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001590874,0.00008002655,0.0001283567,0.0001366255,0.00007165729,0.00006143587,0.0006247597,0.00002546292,0.00003329718],"category_scores_gemma":[0.0003244296,0.00006125715,0.00004437969,0.0007363014,0.00000814581,0.0001953922,0.0005358375,0.0001236293,0.00007369951],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003869819,"about_ca_system_score_gemma":0.00005367307,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007217725,"about_ca_topic_score_gemma":0.0008329595,"domain_scores_codex":[0.9987741,0.0001048078,0.0002990921,0.0003202055,0.0002536366,0.0002481618],"domain_scores_gemma":[0.9990851,0.0002248273,0.00004048558,0.0005105634,0.00008415781,0.00005484715],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004353681,0.00003556462,0.0004731794,0.00001306558,0.00000533242,0.000001072966,0.08316668,0.781934,0.03337628,0.057327,0.00006547209,0.04359804],"study_design_scores_gemma":[0.0001593531,0.00004016919,0.001213752,0.00001952493,0.00000150582,5.372768e-7,0.00279125,0.9939662,0.0006090561,0.001089333,0.00003582345,0.00007355979],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.473327,0.000008135218,0.5246407,0.001030529,0.00006362713,0.0001665492,0.000001394201,0.00006967782,0.0006923268],"genre_scores_gemma":[0.9295412,3.238037e-7,0.06904165,0.000665986,0.00001688968,0.00004463271,0.000001044955,0.000006045658,0.0006821808],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4562142,"threshold_uncertainty_score":0.2497994,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08345484632105207,"score_gpt":0.3630739392796591,"score_spread":0.2796190929586071,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}