{"id":"W3099471123","doi":"10.48550/arxiv.2011.10036","title":"On the Dynamics of Training Attention Models","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Discriminative model; Computer science; Classifier (UML); Artificial intelligence; Embedding; Machine learning; Stochastic gradient descent; Dynamics (music); Task (project management); Set (abstract data type); Block (permutation group theory); Simple (philosophy); Artificial neural network; Natural language processing; Psychology; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001999772,0.00016245,0.0002074328,0.00009411325,0.00007701529,0.00004417489,0.001658884,0.0001336172,0.000006928694],"category_scores_gemma":[0.00002681722,0.0001527644,0.0001718185,0.0002651345,0.00005004096,0.0001716101,0.001160804,0.0004550469,0.00001146272],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001203625,"about_ca_system_score_gemma":0.0001115409,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005357354,"about_ca_topic_score_gemma":0.00001747886,"domain_scores_codex":[0.9988326,0.00009678987,0.0001598862,0.0006426605,0.00009971731,0.0001683221],"domain_scores_gemma":[0.9986045,0.0001208704,0.0002096065,0.0009306511,0.00007288074,0.00006142941],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003113558,0.000008295595,0.00002001411,0.00001385431,0.00001740604,0.0000128447,0.0002650411,0.4275594,0.000006058305,0.5716236,0.0000115134,0.0004588783],"study_design_scores_gemma":[0.00007040361,0.00001385237,0.00003125635,0.00004901625,0.00001296236,4.221603e-7,0.00009505004,0.6417243,0.000006316285,0.357904,0.000002365203,0.00008998348],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1801511,0.00000546311,0.8149149,0.0007951459,0.0002344489,0.0001390006,0.000006513957,0.00008669628,0.003666684],"genre_scores_gemma":[0.9970938,0.00001729876,0.002438181,0.0001285554,0.00002746261,3.988924e-7,0.000005907315,0.000008930494,0.0002794489],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8169427,"threshold_uncertainty_score":0.622955,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1867217059251652,"score_gpt":0.1889230455187379,"score_spread":0.002201339593572688,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}