{"id":"W6891714828","doi":"10.48448/1wdh-0361","title":"Meta-Learning Fast Weight Language Models","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Transformer; Language model; Component (thermodynamics); Key (lock); Artificial neural network; Gradient method","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008010947,0.00164999,0.000756181,0.0006951258,0.000325695,0.001484973,0.002346367,0.001030559,0.01175792],"category_scores_gemma":[0.003482693,0.0006880817,0.0008371255,0.0006648822,0.0003964611,0.003480519,0.001612675,0.002051866,0.007316327],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008386301,"about_ca_system_score_gemma":0.001042589,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004859631,"about_ca_topic_score_gemma":0.01158629,"domain_scores_codex":[0.9994982,0.0001184024,0.00002435987,0.0002029724,0.00009875003,0.00005724678],"domain_scores_gemma":[0.9991423,0.0003091989,0.00005683084,0.0002619124,0.0001825574,0.0000471355],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002253721,0.0001474706,0.0009466636,0.0001913078,0.000175241,0.0001486712,0.00008237869,0.2540824,0.01830051,0.01652444,0.01928074,0.6898949],"study_design_scores_gemma":[0.0000132654,0.00002072266,0.0001028918,0.00001064228,0.00001785369,0.00002706915,0.000008256411,0.9820816,0.005044116,0.01024949,0.00241303,0.00001102636],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02220481,0.0008277071,0.9507864,0.0005349216,0.000224857,0.0000650755,0.000800603,0.01616587,0.008389683],"genre_scores_gemma":[0.5522496,0.0008300598,0.4016569,0.0004853679,0.0002394549,0.0002407163,0.00377101,0.002976073,0.0375509],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01175792,"threshold_uncertainty_score":0.03933412,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04945542335079865,"score_gpt":0.2957450855159473,"score_spread":0.2462896621651487,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}