{"id":"W2949190276","doi":"","title":"On the difficulty of training Recurrent Neural Networks","year":2012,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Neural Networks and Applications","field":"Computer Science","cited_by":140,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Constraint (computer-aided design); Perspective (graphical); Computer science; Artificial neural network; Simple (philosophy); Norm (philosophy); Artificial intelligence; Gradient descent; Mathematical optimization; Algorithm; Applied mathematics; Mathematics; Geometry; Epistemology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006987333,0.001549584,0.00173633,0.0006906053,0.0008311091,0.002058905,0.002243487,0.003567969,0.004669994],"category_scores_gemma":[0.06043579,0.001051125,0.0006641036,0.0008836869,0.002397424,0.007492326,0.003009886,0.005343717,0.001233724],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001028167,"about_ca_system_score_gemma":0.0007986188,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003411078,"about_ca_topic_score_gemma":0.003149687,"domain_scores_codex":[0.9977303,0.001229501,0.0001928869,0.000371912,0.000367095,0.0001083897],"domain_scores_gemma":[0.9636121,0.03262916,0.0007660522,0.001257421,0.001434191,0.0003010606],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00051928,0.0001320966,0.002714365,0.0008141554,0.0001760919,0.0006433103,0.0004575901,0.7165919,0.003192334,0.1206199,0.01569722,0.1384418],"study_design_scores_gemma":[0.00003613351,0.00006017272,0.0002595321,0.00006312785,0.00001575661,0.00007970176,0.00004883151,0.9184043,0.0008707491,0.07880941,0.001334814,0.00001740067],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04733101,0.004844924,0.9262916,0.01007275,0.0004431949,0.0001171081,0.0003011414,0.001489027,0.009109224],"genre_scores_gemma":[0.7266534,0.004783093,0.2514499,0.002446337,0.001632216,0.0005514672,0.0008974455,0.0008649056,0.0107212],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006987333,"threshold_uncertainty_score":0.03695297,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1297134932191586,"score_gpt":0.2002915810523392,"score_spread":0.07057808783318056,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}