{"id":"W2963603213","doi":"","title":"Efficient Exact Gradient Update for training Deep Networks with Very Large Sparse Targets.","year":2015,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal; Canadian Institute for Advanced Research","funders":"","keywords":"Softmax function; Backpropagation; Computer science; Artificial neural network; Speedup; Dimension (graph theory); Algorithm; Deep learning; Computation; Sparse matrix; Gradient descent; Artificial intelligence; Mathematics; Parallel computing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009579326,0.001472142,0.0008987105,0.0004961631,0.0003868302,0.0008620987,0.001547042,0.001305089,0.004253038],"category_scores_gemma":[0.005674418,0.0007954873,0.0005106129,0.000622377,0.0007423551,0.00192982,0.0016442,0.002091172,0.001964164],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001087248,"about_ca_system_score_gemma":0.001499871,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005367836,"about_ca_topic_score_gemma":0.0117366,"domain_scores_codex":[0.9996172,0.00009649808,0.00002584799,0.0000680722,0.000139448,0.00005289408],"domain_scores_gemma":[0.9990003,0.0005510784,0.00007512695,0.0001592043,0.0001619462,0.00005234288],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002591952,0.0001529822,0.001120201,0.00017994,0.00007033969,0.0001210054,0.0001447222,0.630366,0.007162559,0.0246077,0.009217842,0.3265974],"study_design_scores_gemma":[0.00001203568,0.00001408914,0.00004594332,0.000004045244,0.000002819014,0.00001364644,0.000006900281,0.9935228,0.0009301807,0.005004333,0.0004409138,0.000002310549],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007815802,0.0001502194,0.9876605,0.0001849421,0.00004573535,0.00006606536,0.00006353117,0.002758264,0.001254768],"genre_scores_gemma":[0.2395405,0.0002074984,0.7534686,0.0002971583,0.00008127939,0.0003517528,0.0007185646,0.0004974023,0.004837324],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005367836,"threshold_uncertainty_score":0.01422781,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04118044383817406,"score_gpt":0.2430355192364063,"score_spread":0.2018550753982322,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}