{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":749,"total_is_capped":false,"direct_labels_cover":1,"predictions_cover":749,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"fef9bd4f6705","filters":{"topic":"Speech Recognition and Synthesis"}},"results":[{"id":"W2160815625","doi":"10.1109/msp.2012.2205597","title":"Deep Neural Networks for Acoustic Modeling in Speech Recognition: The Shared Views of Four Research Groups","year":2012,"lang":"en","type":"article","venue":"IEEE Signal Processing Magazine","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":10313,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo; University of Toronto","funders":"","keywords":"Hidden Markov model; Speech recognition; Computer science; Mixture model; Artificial neural network; Margin (machine learning); Deep neural networks; Pattern recognition (psychology); Frame (networking); Artificial intelligence; Acoustic model; Gaussian; Speech processing; Machine learning","authors":[{"name":"Geoffrey E. Hinton","is_ca":true},{"name":"Li Deng","is_ca":true},{"name":"Dong Yu","is_ca":false},{"name":"George E. Dahl","is_ca":true},{"name":"Abdelrahman Mohamed","is_ca":true},{"name":"Navdeep Jaitly","is_ca":true},{"name":"Andrew Senior","is_ca":false},{"name":"Vincent Vanhoucke","is_ca":false},{"name":"Patrick Nguyen","is_ca":false},{"name":"Tara N. Sainath","is_ca":false},{"name":"Brian Kingsbury","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2083977889980022,"gpt":0.3405780427773996,"spread":0.1321802537793973,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01315874,0.001442838,0.001326207,0.002396411,0.0008975053,0.00589638,0.00161119,0.003095706,0.001366946],"category_scores_gemma":[0.008225771,0.0007235217,0.0008312339,0.001817255,0.004765124,0.01145104,0.005651735,0.009063808,0.0009994664],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002269783,"about_ca_system_score_gemma":0.002395683,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002025291,"about_ca_topic_score_gemma":0.002222325,"domain_scores_codex":[0.9950436,0.001783625,0.0004169985,0.0006967066,0.001820782,0.0002382268],"domain_scores_gemma":[0.9931086,0.00279588,0.000264526,0.0009318378,0.002332389,0.0005667629],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003367613,0.0001863531,0.002146824,0.00057611,0.0001541525,0.00009470756,0.00194325,0.01846118,0.005896172,0.2474447,0.01615771,0.706602],"study_design_scores_gemma":[0.00008538799,0.0004456573,0.001850493,0.001209946,0.0002201439,0.0003931353,0.00186053,0.1307112,0.02232201,0.5292698,0.3112701,0.000361579],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01926742,0.1446086,0.7136167,0.09515172,0.001905257,0.00009614295,0.0001043898,0.0007403491,0.02450955],"genre_scores_gemma":[0.3152172,0.2327047,0.4113754,0.01139741,0.007566518,0.0003176606,0.0003657571,0.00046198,0.02059343],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01315874,"threshold_uncertainty_score":0.06959087,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2143612262","doi":"10.1109/icassp.2013.6638947","title":"Speech recognition with deep recurrent neural networks","year":2013,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":8837,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Recurrent neural network; Computer science; Connectionism; TIMIT; Speech recognition; Artificial intelligence; Deep learning; Context (archaeology); Benchmark (surveying); Time delay neural network; Artificial neural network; Hidden Markov model; Pattern recognition (psychology)","authors":[{"name":"Alex Graves","is_ca":true},{"name":"Abdelrahman Mohamed","is_ca":true},{"name":"Geoffrey E. Hinton","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02127521644955892,"gpt":0.2153082989889602,"spread":0.1940330825394013,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007586823,0.0006436461,0.0005179227,0.0005044773,0.0001809325,0.0009221419,0.0007264782,0.0007082384,0.00331165],"category_scores_gemma":[0.002312865,0.0003498581,0.0005151977,0.0005718537,0.0002620846,0.001190384,0.0007050282,0.0008824571,0.002779167],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005063308,"about_ca_system_score_gemma":0.0003895355,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00426619,"about_ca_topic_score_gemma":0.007310903,"domain_scores_codex":[0.9993184,0.0001493286,0.00004834553,0.0001720516,0.0002531724,0.00005869945],"domain_scores_gemma":[0.999339,0.0002444432,0.00005612234,0.0001454831,0.0001930563,0.00002175245],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002555691,0.0001000008,0.0009714167,0.0001858028,0.0001741563,0.0001693924,0.0000737193,0.2257454,0.06714167,0.006321207,0.007350397,0.6915113],"study_design_scores_gemma":[0.000008188331,0.00005060496,0.0004709507,0.00001497491,0.00001813379,0.00005218451,0.00001173566,0.9773127,0.01626534,0.003013343,0.002767304,0.00001450159],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04053292,0.001904174,0.944698,0.0003678601,0.0001987479,0.0000498676,0.0005830534,0.006964324,0.004701076],"genre_scores_gemma":[0.6135056,0.001254581,0.370025,0.0002703624,0.000158336,0.0001085123,0.002838765,0.0002847179,0.01155421],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00426619,"threshold_uncertainty_score":0.0110786,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2147768505","doi":"10.1109/tasl.2011.2134090","title":"Context-Dependent Pre-Trained Deep Neural Networks for Large-Vocabulary Speech Recognition","year":2011,"lang":"en","type":"article","venue":"IEEE Transactions on Audio Speech and Language Processing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":3077,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"Microsoft Research; University of Toronto; Microsoft","keywords":"Hidden Markov model; Computer science; Speech recognition; Word error rate; Artificial intelligence; Artificial neural network; Context (archaeology); Mixture model; Deep neural networks; Sentence; Generalization; Phone; Deep learning; Pattern recognition (psychology); Vocabulary; Mathematics","authors":[{"name":"George E. Dahl","is_ca":true},{"name":"Dong Yu","is_ca":false},{"name":"Li Deng","is_ca":false},{"name":"Alex Acero","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02643442671762116,"gpt":0.2531046896995723,"spread":0.2266702629819512,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005402551,0.0007394771,0.0005086479,0.0003133033,0.0002536988,0.0004917633,0.001687372,0.0007769361,0.002501511],"category_scores_gemma":[0.00153387,0.0004941287,0.0005412256,0.0003877989,0.0003055956,0.001191418,0.0007305618,0.001933341,0.00107255],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007601764,"about_ca_system_score_gemma":0.0009559336,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007518476,"about_ca_topic_score_gemma":0.02138541,"domain_scores_codex":[0.9996786,0.00007865491,0.00001941365,0.0001115022,0.00007642763,0.00003542179],"domain_scores_gemma":[0.9995771,0.0001831806,0.00003111435,0.00007630474,0.000110842,0.00002145998],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000216901,0.0001577714,0.001560718,0.0001241886,0.00009306375,0.0001280724,0.00008988482,0.6963683,0.0251768,0.01086662,0.004675195,0.2605426],"study_design_scores_gemma":[0.000002948923,0.00001410471,0.0001425975,0.000003593974,0.000005418377,0.00001479494,0.000002966377,0.9952232,0.002444667,0.001664389,0.0004769564,0.000004298496],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01591547,0.0005193853,0.9801224,0.0001349295,0.00007814744,0.00003435438,0.000239563,0.00167253,0.00128327],"genre_scores_gemma":[0.5965258,0.0007011815,0.3934449,0.0003093461,0.00009234325,0.0002660953,0.0013868,0.0002139815,0.00705944],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007518476,"threshold_uncertainty_score":0.01494944,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2102113734","doi":"","title":"Towards End-To-End Speech Recognition with Recurrent Neural Networks","year":2014,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":1855,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Word error rate; Trigram; Speech recognition; Connectionism; Language model; Recurrent neural network; Artificial intelligence; Lexicon; Artificial neural network; Word (group theory); Time delay neural network; Natural language processing","authors":[{"name":"Alex Graves","is_ca":false},{"name":"Navdeep Jaitly","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02874013249881295,"gpt":0.2433709903689867,"spread":0.2146308578701737,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001251484,0.001157099,0.0007762556,0.000416249,0.0002819698,0.00120865,0.001542653,0.001543361,0.003474607],"category_scores_gemma":[0.002745937,0.0005863355,0.0005047832,0.0003678172,0.0004046608,0.001964444,0.001144379,0.001926122,0.004968637],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004475494,"about_ca_system_score_gemma":0.0004679965,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002284794,"about_ca_topic_score_gemma":0.005185778,"domain_scores_codex":[0.9990326,0.0002481859,0.00006359205,0.0002725974,0.000295546,0.00008744501],"domain_scores_gemma":[0.9989994,0.0003741668,0.00006368425,0.0001562327,0.0003690399,0.00003742312],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005490125,0.0002195632,0.001004686,0.0003356014,0.0001573004,0.0003392654,0.0002534849,0.08600773,0.1780251,0.009024289,0.01180592,0.712278],"study_design_scores_gemma":[0.00002224153,0.000123102,0.0004567922,0.00003591839,0.00003876437,0.0001199631,0.00003743266,0.9391707,0.04825133,0.006219063,0.005495993,0.00002870916],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0106851,0.0004898676,0.9807984,0.0001781628,0.00008990968,0.00003475433,0.0002285846,0.005884231,0.001610942],"genre_scores_gemma":[0.2043234,0.0004793986,0.7818396,0.0003103171,0.000127912,0.0001249469,0.001609392,0.0004352559,0.01074976],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003474607,"threshold_uncertainty_score":0.01162368,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2005708641","doi":"10.1109/asru.2013.6707742","title":"Hybrid speech recognition with Deep Bidirectional LSTM","year":2013,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":1794,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"TIMIT; Computer science; Word error rate; Speech recognition; Leverage (statistics); Hidden Markov model; Artificial neural network; Recurrent neural network; Artificial intelligence; Deep neural networks; Word (group theory); Vocabulary; Acoustic model; Deep learning; Speech processing","authors":[{"name":"Alex Graves","is_ca":true},{"name":"Navdeep Jaitly","is_ca":true},{"name":"Abdelrahman Mohamed","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01839249538427433,"gpt":0.206168135037918,"spread":0.1877756396536437,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007087198,0.0007244415,0.0006105091,0.0005002004,0.0002382685,0.0008719816,0.0009976859,0.0007765848,0.004740981],"category_scores_gemma":[0.001032607,0.00040562,0.000513311,0.0005287207,0.0002291737,0.001396799,0.0008795346,0.0008987238,0.003171199],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005023192,"about_ca_system_score_gemma":0.0004606413,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005112895,"about_ca_topic_score_gemma":0.01006185,"domain_scores_codex":[0.9995766,0.0000819097,0.00002613578,0.0001398884,0.0001209282,0.00005454088],"domain_scores_gemma":[0.9996156,0.0001381731,0.0000241463,0.00007070456,0.0001303614,0.00002097067],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005492383,0.0001879615,0.0008769353,0.0002066932,0.0002151964,0.0002084404,0.0001464562,0.1156329,0.1064973,0.004152012,0.008426363,0.7629004],"study_design_scores_gemma":[0.00001737023,0.00007119899,0.000511224,0.00001341876,0.00003129254,0.00008184098,0.00002960247,0.9671316,0.02629052,0.002837402,0.002958818,0.0000258189],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05921132,0.001228378,0.9147481,0.0003145846,0.0002954403,0.00008470022,0.0009966131,0.01626609,0.006854757],"genre_scores_gemma":[0.6456982,0.0004218286,0.3378663,0.0003298515,0.000102265,0.0001657065,0.001925594,0.0004516861,0.0130386],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005112895,"threshold_uncertainty_score":0.01586014,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1993882792","doi":"10.1109/tasl.2011.2109382","title":"Acoustic Modeling Using Deep Belief Networks","year":2011,"lang":"en","type":"article","venue":"IEEE Transactions on Audio Speech and Language Processing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":1746,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"University of Pennsylvania","keywords":"TIMIT; Hidden Markov model; Discriminative model; Computer science; Artificial intelligence; Pattern recognition (psychology); Deep belief network; Mixture model; Artificial neural network; Speech recognition; Feature (linguistics); Backpropagation; Feature extraction; Markov model; Generative model; Gaussian; Generative grammar; Machine learning; Markov chain","authors":[{"name":"Abdelrahman Mohamed","is_ca":true},{"name":"George E. Dahl","is_ca":true},{"name":"Geoffrey E. Hinton","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03523967748347272,"gpt":0.249923230056549,"spread":0.2146835525730763,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005778009,0.0008502359,0.0007439477,0.0007547314,0.0003251124,0.001316572,0.001527908,0.001127754,0.003588366],"category_scores_gemma":[0.002565627,0.0008324791,0.0007777965,0.0008260459,0.0005630906,0.00156945,0.001062735,0.001887814,0.001492608],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008714126,"about_ca_system_score_gemma":0.0008337053,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01257958,"about_ca_topic_score_gemma":0.01330214,"domain_scores_codex":[0.9996896,0.00007965523,0.00001767339,0.00007048782,0.0001105741,0.00003201232],"domain_scores_gemma":[0.9993027,0.0003858157,0.00006012429,0.00006559764,0.0001549583,0.00003064143],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000238556,0.00001309323,0.0002074249,0.00003131824,0.00002638274,0.00002641532,0.00002036994,0.9595362,0.0007421607,0.007592321,0.0007504394,0.03103003],"study_design_scores_gemma":[0.000001272441,0.000001755705,0.00001970453,0.000002226835,0.000001300314,0.000003277009,0.000001187371,0.9955257,0.0001048448,0.004118285,0.0002186096,0.000001939185],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004071795,0.0004399762,0.9926538,0.0001838752,0.00004359268,0.0000159388,0.0001936221,0.0008661503,0.001531209],"genre_scores_gemma":[0.647759,0.001791405,0.3382541,0.0002449809,0.0001663079,0.0002424901,0.001130731,0.0003693275,0.0100416],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01257958,"threshold_uncertainty_score":0.02501273,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2062227835","doi":"10.1109/icassp.2013.6639346","title":"Improving deep neural networks for LVCSR using rectified linear units and dropout","year":2013,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":1281,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Dropout (neural networks); Artificial neural network; Speech recognition; Artificial intelligence; Sigmoid function; Discriminative model; Deep neural networks; Task (project management); Vocabulary; Deep learning; Mixture model; Hidden Markov model; Time delay neural network; Bayesian probability; Pattern recognition (psychology); Machine learning","authors":[{"name":"George E. Dahl","is_ca":true},{"name":"Tara N. Sainath","is_ca":false},{"name":"Geoffrey E. Hinton","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0442634602867148,"gpt":0.2510945759761163,"spread":0.2068311156894015,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002555288,0.002196582,0.001022506,0.0008457405,0.0006046871,0.0009731093,0.001693571,0.001547885,0.006583173],"category_scores_gemma":[0.005527613,0.0007281979,0.0009515456,0.0007407653,0.0005077788,0.002338533,0.001190327,0.00333959,0.003333926],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001666879,"about_ca_system_score_gemma":0.001470251,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02303117,"about_ca_topic_score_gemma":0.03374742,"domain_scores_codex":[0.999082,0.0001901563,0.00007255267,0.0002367718,0.0003019021,0.0001166822],"domain_scores_gemma":[0.9988064,0.0005776086,0.00006854966,0.0001678975,0.0003395939,0.00003999847],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003462687,0.000230877,0.001197521,0.0002064745,0.0001566436,0.0001689383,0.0001289763,0.3860774,0.02378022,0.004353059,0.009505271,0.5738485],"study_design_scores_gemma":[0.00001970191,0.00007813721,0.0003080294,0.00001791244,0.00002275298,0.00003553672,0.00001440596,0.9854512,0.01079618,0.001231429,0.002008651,0.0000160608],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03877688,0.002205609,0.9407361,0.0004662909,0.0002238933,0.0001262378,0.0004304609,0.01308756,0.003946934],"genre_scores_gemma":[0.4202078,0.001042487,0.5617856,0.000572659,0.0001302954,0.0002781184,0.002304496,0.0009571741,0.01272141],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02303117,"threshold_uncertainty_score":0.04579425,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2747874407","doi":"10.21437/interspeech.2017-1386","title":"Montreal Forced Aligner: Trainable Text-Speech Alignment Using Kaldi","year":2017,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":1124,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Computer science; Speech recognition; Natural language processing; Linguistics; Artificial intelligence","authors":[{"name":"Michael McAuliffe","is_ca":false},{"name":"Michaela Socolof","is_ca":false},{"name":"Sarah Mihuc","is_ca":false},{"name":"Michael Wagner","is_ca":false},{"name":"Morgan Sonderegger","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04401314992965729,"gpt":0.2826153983171813,"spread":0.238602248387524,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001516595,0.00277767,0.001390084,0.001683278,0.001093761,0.001602286,0.002982689,0.001746311,0.04015575],"category_scores_gemma":[0.003513813,0.001203066,0.0009146577,0.001191386,0.0005380725,0.001834674,0.002860261,0.002815411,0.03271436],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001230704,"about_ca_system_score_gemma":0.002998003,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02495923,"about_ca_topic_score_gemma":0.07083222,"domain_scores_codex":[0.9987844,0.0001771878,0.00007479649,0.0005425241,0.0002748267,0.000146321],"domain_scores_gemma":[0.9990196,0.0002922169,0.00004745299,0.000250816,0.0003099108,0.000080104],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009877695,0.0001402019,0.0009038033,0.0004361517,0.0002449155,0.0003274812,0.0001405382,0.01511038,0.06774936,0.004793619,0.09338108,0.8157847],"study_design_scores_gemma":[0.0003450229,0.0004149374,0.00299967,0.00007115441,0.0001696539,0.0004948851,0.0001874056,0.6596141,0.2119787,0.007671501,0.1158386,0.0002143595],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006596823,0.0007776082,0.8189954,0.0001815179,0.0005664086,0.0002779907,0.00539089,0.1621147,0.00509851],"genre_scores_gemma":[0.0815911,0.0004052924,0.8635892,0.0004231085,0.00013866,0.0006317325,0.02319972,0.008349862,0.02167131],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04015575,"threshold_uncertainty_score":0.1343343,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2107638917","doi":"10.1109/tasl.2006.881693","title":"Joint Factor Analysis Versus Eigenchannels in Speaker Recognition","year":2007,"lang":"en","type":"article","venue":"IEEE Transactions on Audio Speech and Language Processing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":709,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Normalization (sociology); NIST; Speaker recognition; Speech recognition; Computer science; Speaker verification; Gaussian; Joint (building); Session (web analytics); Mixture model; Pattern recognition (psychology); Artificial intelligence; Engineering","authors":[{"name":"Patrick Kenny","is_ca":true},{"name":"Gilles Boulianne","is_ca":true},{"name":"Pierre Ouellet","is_ca":true},{"name":"Pierre Dumouchel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0407489263447483,"gpt":0.2811270787352761,"spread":0.2403781523905278,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006414473,0.00113168,0.001011412,0.001350964,0.0004297637,0.002341936,0.0009383538,0.001220211,0.005599441],"category_scores_gemma":[0.02793589,0.0003896536,0.0007188842,0.002223614,0.001345557,0.004267405,0.001442688,0.001141924,0.003294464],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004807131,"about_ca_system_score_gemma":0.0005285912,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002016813,"about_ca_topic_score_gemma":0.003819518,"domain_scores_codex":[0.9903499,0.006358341,0.0004276049,0.0008667157,0.001722268,0.0002752601],"domain_scores_gemma":[0.9796061,0.01522485,0.0006869588,0.002545544,0.001781996,0.0001544833],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003189225,0.000160814,0.003751269,0.000329107,0.0002419329,0.0001004912,0.0002808153,0.06640807,0.02841852,0.03730723,0.009731788,0.8500807],"study_design_scores_gemma":[0.0001123185,0.0004921952,0.006902667,0.00006133983,0.0001251751,0.0003502136,0.0002015589,0.8896384,0.05312815,0.04322869,0.005607787,0.0001513867],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03483377,0.001313969,0.9567072,0.0006322021,0.0001257787,0.0001012657,0.0002709239,0.002097678,0.003917177],"genre_scores_gemma":[0.2928068,0.0009367064,0.7006531,0.0002050965,0.0002074207,0.000219491,0.0005457183,0.0008743763,0.003551295],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006414473,"threshold_uncertainty_score":0.03392339,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2136879537","doi":"10.1109/tasl.2008.925147","title":"A Study of Interspeaker Variability in Speaker Verification","year":2008,"lang":"en","type":"article","venue":"IEEE Transactions on Audio Speech and Language Processing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":591,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"NIST; Hyperparameter; Channel (broadcasting); Computer science; Word error rate; Task (project management); Joint (building); Speech recognition; Statistics; Function (biology); Error analysis; Pattern recognition (psychology); Artificial intelligence; Mathematics; Engineering; Applied mathematics","authors":[{"name":"Patrick Kenny","is_ca":true},{"name":"Pierre Ouellet","is_ca":true},{"name":"Najim Dehak","is_ca":true},{"name":"V. Gupta","is_ca":true},{"name":"Pierre Dumouchel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02538206399158666,"gpt":0.2646189760404997,"spread":0.2392369120489131,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054823,0.0004154686,0.0006011832,0.0004412311,0.0003061552,0.000638011,0.0006007081,0.0006918638,0.0007250534],"category_scores_gemma":[0.02387421,0.0002645716,0.0003439408,0.0005353735,0.0005702113,0.001435221,0.0006767971,0.000871115,0.000184446],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003139252,"about_ca_system_score_gemma":0.0003416269,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001822116,"about_ca_topic_score_gemma":0.00155446,"domain_scores_codex":[0.996472,0.001880463,0.0001188249,0.0006306827,0.0007826362,0.0001155276],"domain_scores_gemma":[0.9732256,0.02333077,0.0008104208,0.00143403,0.001057136,0.0001420266],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00135964,0.0002005469,0.02862355,0.0003395361,0.0005164976,0.0005300751,0.001293984,0.4070306,0.1200332,0.01627464,0.00123835,0.4225594],"study_design_scores_gemma":[0.000009915346,0.0002252119,0.01709648,0.00001737433,0.00004018029,0.0004549482,0.00006745176,0.9520519,0.02612805,0.003075479,0.0007967479,0.00003627046],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.340203,0.001039345,0.6570774,0.0001166878,0.00002067704,0.00002699645,0.00006263301,0.0002962899,0.00115698],"genre_scores_gemma":[0.9454312,0.0002151222,0.05337868,0.00002867628,0.00003387855,0.00001985747,0.0001842088,0.00007755381,0.0006308057],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0054823,"threshold_uncertainty_score":0.02899355,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3167533889","doi":"10.48550/arxiv.2106.04624","title":"SpeechBrain: A General-Purpose Speech Toolkit","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":513,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Sherbrooke; McGill University","funders":"","keywords":"Computer science; Scripting language; Python (programming language); Inference; Architecture; Speech processing; Speech recognition; Speech technology; Natural language processing; Artificial intelligence; Programming language","authors":[{"name":"Mirco Ravanelli","is_ca":false},{"name":"Titouan Parcollet","is_ca":false},{"name":"Peter Plantinga","is_ca":false},{"name":"Aku Rouhe","is_ca":false},{"name":"Samuele Cornell","is_ca":false},{"name":"Loren Lugosch","is_ca":true},{"name":"Cem Subakan","is_ca":false},{"name":"Nauman Dawalatabad","is_ca":false},{"name":"Abdelwahab Heba","is_ca":false},{"name":"Jianyuan Zhong","is_ca":false},{"name":"Ju-Chieh Chou","is_ca":false},{"name":"Sung-Lin Yeh","is_ca":false},{"name":"Szu‐Wei Fu","is_ca":false},{"name":"Chien-Feng Liao","is_ca":false},{"name":"Elena Rastorgueva","is_ca":false},{"name":"François Grondin","is_ca":true},{"name":"William Aris","is_ca":true},{"name":"Hwidong Na","is_ca":false},{"name":"Yan Gao","is_ca":false},{"name":"Renato De Mori","is_ca":true},{"name":"Yoshua Bengio","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09161433237207015,"gpt":0.1896422480167848,"spread":0.09802791564471466,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009344483,0.002090762,0.001035472,0.001182997,0.0005969231,0.001839342,0.00257331,0.001390126,0.0647307],"category_scores_gemma":[0.004827425,0.00114055,0.001072719,0.0006692912,0.0004813892,0.002235929,0.003017886,0.002067142,0.06926867],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004591332,"about_ca_system_score_gemma":0.001289588,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003336742,"about_ca_topic_score_gemma":0.006060784,"domain_scores_codex":[0.9992141,0.00015144,0.00008377009,0.0002117225,0.0002652036,0.00007382525],"domain_scores_gemma":[0.9987925,0.0005580696,0.0000613696,0.0002240085,0.0002697583,0.00009428856],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0008782633,0.000112425,0.001165158,0.002438882,0.000235844,0.0005826215,0.0007605764,0.009395744,0.02851437,0.009969153,0.6665984,0.2793485],"study_design_scores_gemma":[0.0003873793,0.0001994188,0.003524977,0.0004133637,0.0001457205,0.001914993,0.0004215602,0.1808187,0.06135092,0.04327035,0.7071257,0.0004268502],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"software","genre_gemma":"software","genre_scores_codex":[0.003940797,0.0006761044,0.4137305,0.0002890175,0.0004506121,0.0003044497,0.03502375,0.5330475,0.01253735],"genre_scores_gemma":[0.08356533,0.001091829,0.5462098,0.001310982,0.0002940089,0.002313898,0.1625236,0.1578991,0.04479146],"genre_candidate":"software","genre_consensus":"software","teacher_disagreement_score":0.0647307,"threshold_uncertainty_score":0.2165458,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W100623710","doi":"10.1007/3-540-33486-6_6","title":"Neural Probabilistic Language Models","year":2006,"lang":"en","type":"book-chapter","venue":"Studies in fuzziness and soft computing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":491,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Generalization; Sentence; Curse of dimensionality; Artificial intelligence; Word (group theory); Language model; Natural language processing; Probabilistic logic; Representation (politics); Sequence (biology); Set (abstract data type); Linguistics; Mathematics","authors":[{"name":"Yoshua Bengio","is_ca":true},{"name":"Holger Schwenk","is_ca":false},{"name":"Jean-Sébastien Senécal","is_ca":true},{"name":"Fréderic Morin","is_ca":true},{"name":"Jean‐Luc Gauvain","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06406113222153774,"gpt":0.28941119288099,"spread":0.2253500606594523,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000748962,0.0007085752,0.0009071883,0.000767518,0.0005122952,0.001796781,0.001486433,0.001236708,0.008150233],"category_scores_gemma":[0.004634872,0.0007441502,0.0007722611,0.0009652448,0.001039335,0.003397303,0.0009826131,0.002277497,0.002802119],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007161426,"about_ca_system_score_gemma":0.0006534586,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002519765,"about_ca_topic_score_gemma":0.002792685,"domain_scores_codex":[0.9996055,0.0001109167,0.00001995028,0.000111944,0.0001222812,0.00002953754],"domain_scores_gemma":[0.9987695,0.0008431393,0.00005414742,0.0001617543,0.0001423604,0.00002904683],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006603264,0.0000476696,0.0003900486,0.0001349571,0.00008709678,0.00009578526,0.0001698115,0.1777101,0.002021903,0.6350734,0.01354029,0.1706628],"study_design_scores_gemma":[0.000006574206,0.000009971405,0.0001533192,0.00002145357,0.00002054952,0.00007404952,0.00001710162,0.5116266,0.0005952089,0.4794908,0.00796802,0.00001651052],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01158371,0.005934897,0.9454951,0.002386558,0.0004325583,0.00002742846,0.0005538813,0.001415595,0.03217041],"genre_scores_gemma":[0.653699,0.008991632,0.2366909,0.0009063329,0.0009764914,0.0002666358,0.002826261,0.0006974025,0.09494527],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008150233,"threshold_uncertainty_score":0.02726519,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2121415728","doi":"10.1109/tsa.2004.840940","title":"Eigenvoice modeling with sparse training data","year":2005,"lang":"en","type":"article","venue":"IEEE Transactions on Speech and Audio Processing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":472,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Speech recognition; Training set; Set (abstract data type); Adaptation (eye); Limit (mathematics); Training (meteorology); Maximum likelihood; Covariance matrix; Pattern recognition (psychology); Estimation theory; Artificial intelligence; Covariance; Speaker recognition; Algorithm; Mathematics; Statistics","authors":[{"name":"Patrick Kenny","is_ca":true},{"name":"Gilles Boulianne","is_ca":true},{"name":"Pierre Dumouchel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.095794255220844,"gpt":0.2797844726229301,"spread":0.1839902174020861,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007939901,0.0005304951,0.0007143595,0.0004945954,0.0002693826,0.0007678655,0.001352672,0.0008075425,0.001485285],"category_scores_gemma":[0.003208085,0.0005329204,0.0006735871,0.0005643721,0.0005621274,0.001225997,0.001061094,0.001267448,0.000690662],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000251907,"about_ca_system_score_gemma":0.0004872123,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00244556,"about_ca_topic_score_gemma":0.004236388,"domain_scores_codex":[0.9995765,0.0001189921,0.00001632579,0.000101235,0.0001287723,0.00005818656],"domain_scores_gemma":[0.9991251,0.0004424506,0.00008881113,0.0001513066,0.0001627444,0.00002962728],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001051985,0.00005204139,0.000860624,0.00009482021,0.0000609196,0.0001085586,0.000232753,0.8014677,0.01244755,0.03343184,0.002237855,0.1489002],"study_design_scores_gemma":[0.000002632565,0.000007831561,0.0001053101,0.000003516658,0.000002966722,0.00002115068,0.000006456563,0.9912905,0.001047843,0.006963778,0.0005419836,0.000006088376],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005680046,0.00006269001,0.9935176,0.00004362754,0.00001368886,0.000009312174,0.00004555209,0.0001437738,0.0004837273],"genre_scores_gemma":[0.4754205,0.0005498425,0.515599,0.0001857664,0.0001511257,0.0002279883,0.0007995635,0.0002749555,0.006791252],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00244556,"threshold_uncertainty_score":0.004968762,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2188183693","doi":"10.21437/interspeech.2013-744","title":"Exploring convolutional neural network structures and optimization techniques for speech recognition","year":2013,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":388,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"","keywords":"Computer science; Convolutional neural network; Speech recognition; Artificial neural network; Artificial intelligence; Recurrent neural network; Time delay neural network","authors":[{"name":"Ossama Abdel‐Hamid","is_ca":true},{"name":"Li Deng","is_ca":false},{"name":"Dong Yu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09942673034509407,"gpt":0.2529693359682053,"spread":0.1535426056231112,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009021444,0.0009582224,0.0003728415,0.0005512431,0.0001920943,0.0005013238,0.0006422909,0.0006804772,0.001843267],"category_scores_gemma":[0.002412337,0.0005313866,0.000468861,0.0006202165,0.0004477463,0.001317611,0.0005640354,0.0009942306,0.0003978361],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009688829,"about_ca_system_score_gemma":0.0008418691,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007657456,"about_ca_topic_score_gemma":0.01381113,"domain_scores_codex":[0.9997811,0.0000639587,0.00001345837,0.00005328387,0.00006050345,0.00002769262],"domain_scores_gemma":[0.9995965,0.0002541492,0.00003863204,0.00003822003,0.00006206777,0.00001041014],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006713961,0.00004269534,0.0005876063,0.0001131815,0.00007328575,0.0000454584,0.00003169056,0.8388044,0.009638952,0.01636015,0.00115408,0.1330814],"study_design_scores_gemma":[0.000002937505,0.00001097456,0.0000760441,0.000005497726,0.000005459241,0.000004850875,0.000002612001,0.9949066,0.001500884,0.003050832,0.000430791,0.000002486966],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02775985,0.001907346,0.965208,0.0004494195,0.00004555968,0.00003615155,0.00008813597,0.0008235237,0.003682073],"genre_scores_gemma":[0.5340862,0.002624897,0.4563635,0.0002138489,0.00008332929,0.0001286163,0.0003295413,0.00022228,0.005947806],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007657456,"threshold_uncertainty_score":0.01522577,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2963403664","doi":"10.21437/interspeech.2016-1446","title":"Towards End-to-End Speech Recognition with Deep Convolutional Neural Networks","year":2016,"lang":"en","type":"preprint","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":341,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Canadian Institute for Advanced Research; Université de Montréal","funders":"","keywords":"End-to-end principle; Computer science; Speech recognition; Convolutional neural network; Deep learning; Artificial intelligence","authors":[{"name":"Ying Zhang","is_ca":true},{"name":"Mohammad Pezeshki","is_ca":true},{"name":"Philémon Brakel","is_ca":true},{"name":"Saizheng Zhang","is_ca":true},{"name":"César Laurent","is_ca":true},{"name":"Yoshua Bengio","is_ca":true},{"name":"Aaron Courville","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03436162498600204,"gpt":0.2491030066026894,"spread":0.2147413816166874,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009233562,0.001382523,0.0007525792,0.0004814634,0.0003201676,0.001071959,0.001736085,0.001614629,0.003091725],"category_scores_gemma":[0.002103507,0.0004994198,0.000519193,0.0004705319,0.0004734243,0.00177488,0.001292192,0.001964099,0.003504051],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007616364,"about_ca_system_score_gemma":0.0008674857,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004642928,"about_ca_topic_score_gemma":0.009980625,"domain_scores_codex":[0.9992722,0.0001433424,0.00004161678,0.0002466746,0.0002071808,0.00008907293],"domain_scores_gemma":[0.999126,0.000243215,0.00006001582,0.0002261375,0.000296323,0.00004828269],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004493821,0.0002997171,0.001238209,0.0002453565,0.00008810541,0.000273938,0.0001594382,0.1179103,0.1040898,0.01459995,0.0142164,0.7464294],"study_design_scores_gemma":[0.00001033993,0.00006413283,0.0003329234,0.00001810489,0.00001491401,0.0000592811,0.00002508425,0.9489625,0.03842642,0.007799723,0.004271243,0.00001536359],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.009250979,0.0003119954,0.9830112,0.0001551842,0.00006832998,0.00003954097,0.0002558733,0.005293387,0.001613373],"genre_scores_gemma":[0.2262653,0.0003971161,0.7613549,0.0003421691,0.00007610745,0.0001259369,0.002115375,0.0003463506,0.008976784],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004642928,"threshold_uncertainty_score":0.01034284,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2901997113","doi":"","title":"Char2Wav: End-to-End Speech Synthesis","year":2017,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":335,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Institut National de la Recherche Scientifique; Université de Montréal","funders":"","keywords":"End-to-end principle; Computer science; Speech synthesis; Speech recognition; Artificial intelligence","authors":[{"name":"Jose Sotelo","is_ca":true},{"name":"Soroush Mehri","is_ca":false},{"name":"Kundan Kumar","is_ca":false},{"name":"João Felipe Santos","is_ca":true},{"name":"Kyle Kastner","is_ca":true},{"name":"Aaron Courville","is_ca":true},{"name":"Yoshua Bengio","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09295328918588806,"gpt":0.3624233461629727,"spread":0.2694700569770846,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007696985,0.002380631,0.001344818,0.0009871474,0.0005950771,0.001838972,0.002063848,0.001678493,0.05297149],"category_scores_gemma":[0.001656132,0.0007843404,0.001027733,0.0005160046,0.00041587,0.001226423,0.003241386,0.001686798,0.03085852],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004670813,"about_ca_system_score_gemma":0.0007758544,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001785834,"about_ca_topic_score_gemma":0.003682166,"domain_scores_codex":[0.9991267,0.0001022481,0.00004285313,0.0002398504,0.0003574851,0.0001309111],"domain_scores_gemma":[0.999571,0.0001198154,0.00001534092,0.0001375682,0.0001191975,0.00003703954],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002556513,0.0003631441,0.0007032029,0.0005292566,0.0003034805,0.0007676902,0.0002095968,0.01737269,0.1290953,0.007819952,0.1241372,0.716142],"study_design_scores_gemma":[0.0005033736,0.0005182599,0.001299264,0.00008177346,0.00009213998,0.0008094392,0.0001558805,0.5206481,0.3414293,0.01542711,0.1188533,0.0001819944],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006284946,0.0002746154,0.7827226,0.00009152506,0.0005288473,0.0002219966,0.003220903,0.198645,0.008009537],"genre_scores_gemma":[0.1706071,0.0003033589,0.7261784,0.0005675092,0.0002184688,0.001055154,0.02347433,0.02619944,0.05139634],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.05297149,"threshold_uncertainty_score":0.1772073,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2099621636","doi":"","title":"Vocal Tract Length Perturbation (VTLP) improves speech recognition","year":2013,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":335,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Speech recognition; Vocal tract; Normalization (sociology); Convolutional neural network; Spectrogram; Artificial neural network; Test set; Dynamic time warping; Word error rate; TIMIT; Image warping; Artificial intelligence; Pattern recognition (psychology); Hidden Markov model","authors":[{"name":"Navdeep Jaitly","is_ca":true},{"name":"Ella Hinton","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0241960568558066,"gpt":0.2242876426069468,"spread":0.2000915857511402,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001178324,0.001370072,0.0007108885,0.0006693693,0.0002390832,0.0007533395,0.0007133445,0.0006400393,0.003504599],"category_scores_gemma":[0.004030154,0.0002854028,0.0006336476,0.0006386145,0.0004179307,0.001608846,0.001291716,0.001057808,0.00349553],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003566546,"about_ca_system_score_gemma":0.000358993,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001757143,"about_ca_topic_score_gemma":0.002692425,"domain_scores_codex":[0.9991198,0.0002002084,0.00006063042,0.0002937511,0.0002581192,0.00006748896],"domain_scores_gemma":[0.9984262,0.0006635634,0.0001282356,0.000511012,0.0002169565,0.00005407643],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0005605828,0.0001678478,0.001957412,0.0002171746,0.00008743563,0.00016233,0.00009865159,0.02888742,0.1667939,0.000927612,0.004326576,0.7958131],"study_design_scores_gemma":[0.00006164322,0.001360471,0.01839742,0.00008504881,0.0001884989,0.0009634881,0.0001526485,0.637709,0.3128448,0.004387265,0.02372607,0.0001236396],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2150174,0.002991399,0.7426578,0.0005482757,0.0006902863,0.0001422532,0.001915378,0.02946858,0.006568536],"genre_scores_gemma":[0.6597602,0.0009818404,0.3210864,0.0002945862,0.0002438601,0.0001836959,0.007304374,0.00139063,0.008754312],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003504599,"threshold_uncertainty_score":0.01172405,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2103359087","doi":"","title":"Phone Recognition with the Mean-Covariance Restricted Boltzmann Machine","year":2010,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":275,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"TIMIT; Boltzmann machine; Computer science; Covariance; Hidden Markov model; Restricted Boltzmann machine; Speech recognition; Mixture model; Artificial intelligence; Covariance matrix; Pattern recognition (psychology); Algorithm; Artificial neural network; Mathematics; Statistics","authors":[{"name":"George E. Dahl","is_ca":true},{"name":"Marc’Aurelio Ranzato","is_ca":true},{"name":"Abdelrahman Mohamed","is_ca":true},{"name":"Geoffrey E. Hinton","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01803519807988988,"gpt":0.2190759133437164,"spread":0.2010407152638265,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006809735,0.0006142377,0.0007938077,0.0003526073,0.0002523087,0.0007589944,0.001099323,0.001086935,0.002282931],"category_scores_gemma":[0.002781232,0.0005833997,0.000809923,0.0005448192,0.0003941318,0.001248036,0.0009273969,0.001787455,0.001751536],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005120158,"about_ca_system_score_gemma":0.0006979938,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003424067,"about_ca_topic_score_gemma":0.004383217,"domain_scores_codex":[0.9995027,0.0001498032,0.00002743543,0.0001415114,0.0001349366,0.00004367648],"domain_scores_gemma":[0.999511,0.0002426678,0.00003762197,0.0001010558,0.00008655722,0.00002109487],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002228583,0.00007744563,0.001221737,0.00009529945,0.0001477671,0.0001022772,0.00008673325,0.6280969,0.01292071,0.01443225,0.005677124,0.3369189],"study_design_scores_gemma":[0.000004889882,0.00000939156,0.0001225795,0.000003374458,0.000004234852,0.00002138438,0.00000325442,0.9909071,0.001673283,0.006726068,0.000517293,0.000007170093],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.009584766,0.0002790526,0.9864442,0.0001767341,0.00006744931,0.00002227352,0.0001399126,0.001993454,0.00129216],"genre_scores_gemma":[0.4656738,0.00043217,0.5261123,0.0002789912,0.0001101974,0.0002288471,0.000758826,0.000276915,0.006127978],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003424067,"threshold_uncertainty_score":0.007637143,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2172097686","doi":"10.1109/icassp.2012.6288863","title":"Understanding how Deep Belief Networks perform acoustic modelling","year":2012,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":270,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Deep belief network; TIMIT; Computer science; Hidden Markov model; Artificial intelligence; Artificial neural network; Similarity (geometry); Feature (linguistics); Pattern recognition (psychology); Speech recognition; Feature vector; Visualization; Representation (politics)","authors":[{"name":"Abdelrahman Mohamed","is_ca":true},{"name":"Geoffrey E. Hinton","is_ca":true},{"name":"Gerald Penn","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1542638219167346,"gpt":0.2391786981223847,"spread":0.08491487620565008,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001687313,0.0007813609,0.0004277925,0.000724728,0.0004480001,0.003367307,0.0012641,0.001427115,0.003534873],"category_scores_gemma":[0.01084562,0.0007498211,0.0005334429,0.0006587364,0.001022219,0.005933099,0.001097015,0.002342584,0.0008149823],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001231862,"about_ca_system_score_gemma":0.0009341554,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01372974,"about_ca_topic_score_gemma":0.009991081,"domain_scores_codex":[0.9994228,0.0002515773,0.00002638164,0.00008848029,0.0001333953,0.00007738709],"domain_scores_gemma":[0.9976507,0.00161869,0.0001254587,0.0001904199,0.0003214082,0.00009327093],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006522417,0.00004182141,0.002295271,0.0001485242,0.00009920784,0.0001109725,0.0004306845,0.5040702,0.001929257,0.4055953,0.002860168,0.08235339],"study_design_scores_gemma":[0.000005367444,0.000009629904,0.0002320043,0.00003057209,0.000009293662,0.00001713316,0.00005462345,0.7638219,0.0006656916,0.2334282,0.001713624,0.00001190787],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02090118,0.001131864,0.9638004,0.002627853,0.00009725292,0.0000294571,0.000190441,0.0005428229,0.01067876],"genre_scores_gemma":[0.7642179,0.002722459,0.2245065,0.000345097,0.0001084119,0.00009391319,0.0002590599,0.0001805796,0.007566052],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01372974,"threshold_uncertainty_score":0.02729964,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1269046860","doi":"10.1201/9781482276237","title":"Speech Processing","year":2018,"lang":"en","type":"book","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":239,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University; Université du Québec à Montréal","funders":"","keywords":"Speech recognition; Computer science; Natural language processing","authors":[{"name":"Li Deng","is_ca":false},{"name":"Douglas O’Shaughnessy","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0272552053198968,"gpt":0.2467195160514359,"spread":0.2194643107315391,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005027199,0.001187617,0.0007889204,0.001465113,0.0007291888,0.002891493,0.001057216,0.001409587,0.09150737],"category_scores_gemma":[0.001355866,0.0003562292,0.0005401124,0.001484932,0.0006441026,0.001749733,0.001319273,0.001414085,0.1028171],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005313615,"about_ca_system_score_gemma":0.0007077574,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006980454,"about_ca_topic_score_gemma":0.0007944025,"domain_scores_codex":[0.9993419,0.00005868913,0.00004714222,0.0001430137,0.0003785353,0.00003071712],"domain_scores_gemma":[0.999397,0.0001198702,0.00002185626,0.000106865,0.0003200883,0.00003429175],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0000597232,0.00003489662,0.00009179264,0.0007073423,0.00003256378,0.0001858858,0.0001458985,0.00120292,0.01041961,0.02835307,0.2248123,0.733954],"study_design_scores_gemma":[0.000007894409,0.00004631698,0.0003728984,0.0002390932,0.0000116024,0.0006755522,0.00007051687,0.001894112,0.003783353,0.01541881,0.977459,0.00002089225],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.003191577,0.09637336,0.2814465,0.004751036,0.01309894,0.0004351291,0.003076676,0.006931006,0.5906959],"genre_scores_gemma":[0.02151054,0.04494285,0.07578741,0.002558302,0.00391876,0.0003014208,0.004517754,0.001185893,0.8452771],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.09150737,"threshold_uncertainty_score":0.3061227,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2132037657","doi":"10.1109/icassp.2011.5947700","title":"Learning a better representation of speech soundwaves using restricted boltzmann machines","year":2011,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":229,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Cepstrum; Speech recognition; Computer science; Mel-frequency cepstrum; Linear predictive coding; Boltzmann machine; Restricted Boltzmann machine; Representation (politics); Artificial intelligence; Speech coding; Pattern recognition (psychology); Artificial neural network; Feature extraction","authors":[{"name":"Navdeep Jaitly","is_ca":true},{"name":"Geoffrey E. Hinton","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07598615674813156,"gpt":0.2849837548539779,"spread":0.2089975981058464,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008940009,0.0005685542,0.001037107,0.0002944562,0.0002058423,0.0008745331,0.000932127,0.001056824,0.001506037],"category_scores_gemma":[0.003733619,0.0005379779,0.000874279,0.0003112698,0.0005227129,0.002074662,0.0006683923,0.001486319,0.0006135412],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004668699,"about_ca_system_score_gemma":0.0004740619,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001633349,"about_ca_topic_score_gemma":0.001988678,"domain_scores_codex":[0.9996507,0.0001526049,0.00002289269,0.00008902267,0.00005119025,0.00003354511],"domain_scores_gemma":[0.999222,0.0004759167,0.00005761803,0.0001267739,0.00008960974,0.00002807551],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000977109,0.00007390131,0.0007083044,0.00007647688,0.00007514495,0.00005862439,0.00008421391,0.9089369,0.01457294,0.01307112,0.0007550344,0.06148966],"study_design_scores_gemma":[0.000004070934,0.00000823611,0.00004052119,0.000001865207,0.000002651707,0.000005100698,0.000002701754,0.9963164,0.0005059255,0.003023939,0.00008406675,0.000004435276],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04029168,0.0002020041,0.9581484,0.0002651043,0.00004375783,0.00002254106,0.00005523728,0.0004590312,0.0005121914],"genre_scores_gemma":[0.7289907,0.0003438052,0.2665318,0.0002602578,0.00007845676,0.0001808593,0.0002836749,0.0001758049,0.003154685],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.001633349,"threshold_uncertainty_score":0.005038202,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1556470778","doi":"","title":"Prosodylab-aligner: A tool for forced alignment of laboratory speech","year":2011,"lang":"en","type":"article","venue":"Canadian acoustics","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":228,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":true,"about_ca":false},"ca_institutions":"McGill University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Computer science; Hidden Markov model; Unix; Scripting language; Speech recognition; Operating system; Process (computing); Computer graphics (images); Software","authors":[{"name":"Kyle Gorman","is_ca":false},{"name":"Jonathan Howell","is_ca":true},{"name":"Michael Wagner","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02785983580016489,"gpt":0.2188774394414399,"spread":0.191017603641275,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003111311,0.00259157,0.001448016,0.002929933,0.001464125,0.002261287,0.002815969,0.001487362,0.1918505],"category_scores_gemma":[0.01014029,0.001673521,0.001187332,0.001662668,0.0005538366,0.003020824,0.003871529,0.002714864,0.09114804],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004512816,"about_ca_system_score_gemma":0.001130962,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001407068,"about_ca_topic_score_gemma":0.002432077,"domain_scores_codex":[0.9981866,0.000417421,0.0001908594,0.0006257382,0.0004517428,0.0001275867],"domain_scores_gemma":[0.99714,0.001305124,0.0002225532,0.0006153747,0.0005451412,0.0001718197],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009072715,0.0001756455,0.00124028,0.001020296,0.0001946588,0.0006289693,0.0006893649,0.003564544,0.05151069,0.005901587,0.3831896,0.5509771],"study_design_scores_gemma":[0.0005686095,0.0004686841,0.008583353,0.0003318056,0.0001490322,0.00260817,0.0006903021,0.1510601,0.1311238,0.02812154,0.6757885,0.000506062],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.00273416,0.0002417648,0.7582992,0.0001352607,0.0004247602,0.0002584319,0.0194263,0.2131918,0.005288276],"genre_scores_gemma":[0.02883366,0.0001916608,0.8366712,0.000233806,0.0001885696,0.001614872,0.05279806,0.06843407,0.01103428],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.1918505,"threshold_uncertainty_score":0.6418037,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2084514013","doi":"10.1109/msp.2009.932166","title":"Developments and directions in speech recognition and understanding, Part 1 [DSP Education]","year":2009,"lang":"en","type":"article","venue":"IEEE Signal Processing Magazine","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":227,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Institut National de la Recherche Scientifique","funders":"","keywords":"Computer science; Set (abstract data type); Focus (optics); Field (mathematics); Data science; Speech processing; Digital signal processing; Speech recognition","authors":[{"name":"James Baker","is_ca":false},{"name":"Li Deng","is_ca":false},{"name":"James Glass","is_ca":false},{"name":"Sanjeev Khudanpur","is_ca":false},{"name":"Chin-hui Lee","is_ca":false},{"name":"N. Morgan","is_ca":false},{"name":"Douglas O’Shaughnessy","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06475663745855617,"gpt":0.2729375041776006,"spread":0.2081808667190444,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004625231,0.0007129897,0.0004873152,0.002981291,0.000976964,0.005071448,0.00110093,0.003010518,0.01695691],"category_scores_gemma":[0.006946479,0.0004813585,0.0003609601,0.002932438,0.002892744,0.006117096,0.001068062,0.002957051,0.009398795],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001550431,"about_ca_system_score_gemma":0.003303463,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002330675,"about_ca_topic_score_gemma":0.002497551,"domain_scores_codex":[0.9985941,0.0004788737,0.0001804451,0.0002101593,0.0004519026,0.0000844902],"domain_scores_gemma":[0.9942631,0.002873027,0.0002012516,0.0003961581,0.001963665,0.000302805],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004094408,0.0001006806,0.0004214667,0.0009658958,0.00001192125,0.00009566929,0.0006281912,0.0008553021,0.002614999,0.07902663,0.08620545,0.829033],"study_design_scores_gemma":[0.000009140297,0.000163317,0.001471986,0.001158825,0.00001613581,0.0006456958,0.0006135394,0.002703054,0.00253674,0.07608259,0.9145492,0.00004971601],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.005079667,0.5004536,0.1824945,0.08110753,0.01915843,0.00018318,0.00029634,0.00126484,0.2099619],"genre_scores_gemma":[0.0649402,0.6139265,0.1393798,0.009595611,0.01950357,0.0002513461,0.0007643567,0.0005240532,0.1511145],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.01695691,"threshold_uncertainty_score":0.05672646,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2964187693","doi":"10.1109/icassp.2018.8462688","title":"Deep Residual Learning for Small-Footprint Keyword Spotting","year":2018,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":227,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Keyword spotting; Computer science; Residual; Convolutional neural network; Deep learning; Footprint; Memory footprint; Benchmark (surveying); Artificial intelligence; Spotting; Machine learning; Speech recognition; Algorithm; Cartography","authors":[{"name":"Raphael Tang","is_ca":true},{"name":"Jimmy Lin","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04154259823274774,"gpt":0.2640082849263389,"spread":0.2224656866935912,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009271089,0.001211597,0.0009383426,0.0006293041,0.0002485496,0.0008805081,0.001594987,0.0009658507,0.005116117],"category_scores_gemma":[0.004365517,0.0002782261,0.0005212247,0.0006650115,0.0004252594,0.00194544,0.001230091,0.001292604,0.002987074],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005349763,"about_ca_system_score_gemma":0.0009005874,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006148698,"about_ca_topic_score_gemma":0.01027691,"domain_scores_codex":[0.999527,0.00008669947,0.00002923707,0.0001519435,0.0001301785,0.00007502254],"domain_scores_gemma":[0.9990049,0.0004624407,0.0000688508,0.0001983299,0.0001919199,0.00007357394],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001412513,0.0003713434,0.001429065,0.0004720921,0.0001503538,0.0003329726,0.0001600933,0.1711189,0.08666302,0.003891754,0.01325699,0.7207408],"study_design_scores_gemma":[0.00004602599,0.0002160127,0.0003663996,0.00001519379,0.00002649989,0.00009647236,0.00004790772,0.9521105,0.04153012,0.002597784,0.002926918,0.00002015509],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1465676,0.002288071,0.8134872,0.0005081226,0.0002911112,0.0001431439,0.001714618,0.02908599,0.005914182],"genre_scores_gemma":[0.6420081,0.0006181247,0.3420345,0.0003306093,0.0001186333,0.0001209962,0.003618007,0.001109143,0.01004193],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006148698,"threshold_uncertainty_score":0.01711506,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2112582577","doi":"10.1109/tasl.2007.894527","title":"Speaker and Session Variability in GMM-Based Speaker Verification","year":2007,"lang":"en","type":"article","venue":"IEEE Transactions on Audio Speech and Language Processing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":227,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"NIST; Computer science; Speaker recognition; Speech recognition; Session (web analytics); Set (abstract data type); Statistic; Pattern recognition (psychology); Artificial intelligence; Speaker diarisation; Factor (programming language); Feature (linguistics); Statistics; Mathematics; Linguistics","authors":[{"name":"Patrick Kenny","is_ca":true},{"name":"Gilles Boulianne","is_ca":true},{"name":"Pierre Ouellet","is_ca":true},{"name":"Pierre Dumouchel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01295308220622218,"gpt":0.2605477096187157,"spread":0.2475946274124935,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004501142,0.0005478809,0.0008487874,0.000860945,0.0004222172,0.0006467479,0.0006530439,0.0008323531,0.001260295],"category_scores_gemma":[0.009684813,0.0003827809,0.0006325946,0.0006905982,0.0005319506,0.001319716,0.000940619,0.0007773634,0.0008870452],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004924235,"about_ca_system_score_gemma":0.0005930165,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003372102,"about_ca_topic_score_gemma":0.005052742,"domain_scores_codex":[0.9967769,0.001748424,0.0001252059,0.0004892467,0.0006760841,0.0001841329],"domain_scores_gemma":[0.9956408,0.003191883,0.0001873573,0.0005092053,0.0004079868,0.00006278576],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002274929,0.0001367568,0.01721282,0.000288425,0.0003021708,0.0002754008,0.0006859694,0.1772145,0.07318453,0.01450597,0.004049864,0.7098687],"study_design_scores_gemma":[0.00003011217,0.0002488515,0.01849447,0.00004383827,0.0001463443,0.0005832875,0.0001297681,0.9142366,0.0552182,0.007438916,0.003343967,0.00008575443],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1092454,0.001012503,0.8849971,0.000140194,0.00009500627,0.00007472569,0.0002906504,0.001854645,0.002289675],"genre_scores_gemma":[0.8162435,0.0003650576,0.1804785,0.00007132749,0.0001017503,0.0001106545,0.0004947974,0.0002683136,0.001866066],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004501142,"threshold_uncertainty_score":0.02380466,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1993409002","doi":"10.1109/icassp.2013.6639211","title":"Fast speaker adaptation of hybrid NN/HMM model for speech recognition based on discriminative learning of speaker code","year":2013,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":224,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"","keywords":"Speech recognition; Computer science; TIMIT; Hidden Markov model; Speaker diarisation; Adaptation (eye); Speaker recognition; Discriminative model; Artificial intelligence; Artificial neural network; Pattern recognition (psychology)","authors":[{"name":"Ossama Abdel‐Hamid","is_ca":true},{"name":"Hui Jiang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06434906411656907,"gpt":0.2616561921307978,"spread":0.1973071280142288,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007173984,0.0005572086,0.0006756784,0.0003546772,0.0002526895,0.0003384081,0.0008831628,0.00054605,0.001618434],"category_scores_gemma":[0.001032961,0.0003977036,0.0007522022,0.0003315387,0.0002787914,0.0007545213,0.0006465073,0.001230205,0.001293924],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002910008,"about_ca_system_score_gemma":0.0003838051,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003918859,"about_ca_topic_score_gemma":0.005495124,"domain_scores_codex":[0.9994611,0.000120809,0.00002509035,0.0001831385,0.0001673674,0.00004235755],"domain_scores_gemma":[0.9995897,0.0001702462,0.00002389457,0.00008479937,0.0001121753,0.00001908567],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003275498,0.0001329216,0.002054015,0.0001536646,0.0002459878,0.0001513051,0.0001831721,0.1888481,0.09131563,0.003854762,0.003196486,0.7095364],"study_design_scores_gemma":[0.000006853863,0.00003803777,0.0009376568,0.000005043798,0.00002984372,0.0001297002,0.000009536596,0.9847551,0.0109592,0.001080411,0.002025506,0.00002307743],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.009073867,0.0002911855,0.9887938,0.00003365497,0.00006507297,0.00002183344,0.00003584774,0.001051663,0.0006331149],"genre_scores_gemma":[0.4050944,0.0006824751,0.5842597,0.0002089152,0.000148785,0.0002151463,0.0005855305,0.000392595,0.008412408],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003918859,"threshold_uncertainty_score":0.007792115,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2141778357","doi":"10.21437/interspeech.2010-304","title":"Investigation of full-sequence training of deep belief networks for speech recognition","year":2010,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":213,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Training (meteorology); Speech recognition; Sequence (biology); Artificial intelligence; Natural language processing","authors":[{"name":"Abdelrahman Mohamed","is_ca":true},{"name":"Dong Yu","is_ca":false},{"name":"Li Deng","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08245940770069941,"gpt":0.2678514967899197,"spread":0.1853920890892203,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003424853,0.0008575534,0.000715826,0.0003372819,0.0002434749,0.0006367402,0.001068128,0.001009064,0.001728141],"category_scores_gemma":[0.0111438,0.0007569243,0.0003280744,0.000391651,0.0005965377,0.0022943,0.0008666596,0.001541505,0.0002409131],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008890731,"about_ca_system_score_gemma":0.001300156,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004326206,"about_ca_topic_score_gemma":0.005410181,"domain_scores_codex":[0.9993008,0.000336212,0.00003115105,0.0001021756,0.0001737746,0.00005586836],"domain_scores_gemma":[0.9963666,0.00272238,0.0001111896,0.000231034,0.0004962926,0.00007254232],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000154091,0.0001296471,0.0009672606,0.00009271265,0.00004477189,0.00004481435,0.00006575132,0.8446046,0.005295548,0.01446465,0.0004189946,0.1337172],"study_design_scores_gemma":[0.00000259554,0.00002051924,0.00004454263,0.00000306322,0.00000189997,0.000005084686,0.00000244814,0.9981186,0.0007589585,0.0009558877,0.00008505267,0.000001300215],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07590638,0.001365495,0.9182515,0.0005409165,0.00003292864,0.00005970755,0.00004095839,0.0004401691,0.003362067],"genre_scores_gemma":[0.7606762,0.0008768535,0.2356537,0.0001608383,0.00004367312,0.0001207561,0.0001235149,0.0000793568,0.002265058],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004326206,"threshold_uncertainty_score":0.0181126,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2064364374","doi":"10.1109/icassp.2013.6639151","title":"PLDA for speaker verification with utterances of arbitrary duration","year":2013,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":205,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"NIST; Computer science; Speaker verification; Covariance; Speech recognition; Classifier (UML); Speaker recognition; Duration (music); Speech processing; Artificial intelligence; Pattern recognition (psychology); Mathematics; Statistics","authors":[{"name":"Patrick Kenny","is_ca":true},{"name":"Themos Stafylakis","is_ca":true},{"name":"Pierre Ouellet","is_ca":true},{"name":"Jahangir Alam","is_ca":true},{"name":"Pierre Dumouchel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01382047087280255,"gpt":0.2059364721303044,"spread":0.1921160012575018,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002621326,0.000799861,0.0007299178,0.0005750829,0.0006252391,0.0009247653,0.0008782249,0.0007942743,0.003551476],"category_scores_gemma":[0.009210196,0.0004265615,0.0005825878,0.0005432915,0.0005697971,0.001122695,0.001135665,0.001765402,0.002220932],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007093692,"about_ca_system_score_gemma":0.0009226816,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003946471,"about_ca_topic_score_gemma":0.005823688,"domain_scores_codex":[0.9979266,0.0007918924,0.0001430077,0.0005491551,0.0004834637,0.000105908],"domain_scores_gemma":[0.9968975,0.001653619,0.0002533112,0.0005496196,0.0005834692,0.00006247652],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009078489,0.0001119133,0.001617636,0.0003410684,0.0001703163,0.0004035629,0.000291889,0.1185311,0.09448039,0.01332946,0.006092647,0.7637222],"study_design_scores_gemma":[0.00002356456,0.0001358218,0.002121814,0.00003442828,0.00002185556,0.0001812986,0.00004277773,0.963241,0.02232606,0.005764553,0.006065483,0.00004125232],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.009969063,0.0004580731,0.9862186,0.0000891985,0.00005691721,0.00006156942,0.0002344476,0.001966746,0.0009452955],"genre_scores_gemma":[0.2373239,0.0003891306,0.7568973,0.0001235269,0.00006943261,0.0003551854,0.0009422562,0.0003849285,0.003514414],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003946471,"threshold_uncertainty_score":0.01386303,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1990505856","doi":"10.1109/icassp.2014.6854321","title":"Deep mixture density networks for acoustic modeling in statistical parametric speech synthesis","year":2014,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":193,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Concordia University","funders":"","keywords":"Naturalness; Parametric statistics; Computer science; Speech synthesis; Artificial neural network; Speech recognition; Probability density function; Layer (electronics); Deep neural networks; Mixture model; Statistical model; Artificial intelligence; Mathematics; Statistics","authors":[{"name":"Heiga Zen","is_ca":true},{"name":"Andrew Senior","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02765918503457674,"gpt":0.2498129052068847,"spread":0.222153720172308,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009763904,0.0005984272,0.0005053735,0.0004332027,0.0002254138,0.00062724,0.0006745887,0.0006529285,0.001347953],"category_scores_gemma":[0.002251507,0.000584069,0.000564087,0.0004328518,0.0004081733,0.001059804,0.0007288036,0.001347438,0.0004475437],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007021373,"about_ca_system_score_gemma":0.0007449111,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006625595,"about_ca_topic_score_gemma":0.006794692,"domain_scores_codex":[0.9997041,0.0001101061,0.00001666134,0.00005279471,0.00009408695,0.00002227701],"domain_scores_gemma":[0.9993929,0.0004186952,0.0000367294,0.00003448677,0.0001017542,0.00001553073],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001068262,0.00003431027,0.0004733988,0.00009396895,0.00006007013,0.00004983814,0.00006703641,0.85167,0.007836065,0.01479439,0.0006794318,0.1241346],"study_design_scores_gemma":[0.000001138213,0.000004683503,0.00005167469,0.000003102339,0.000003499996,0.000005706825,0.000001805737,0.9971099,0.0007094508,0.001871714,0.000234238,0.000003134621],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007602325,0.0006257988,0.9906545,0.00009169773,0.00002449622,0.00001095646,0.00003852625,0.0003634964,0.0005882285],"genre_scores_gemma":[0.6542566,0.001739941,0.3381546,0.0001201005,0.00007810704,0.0001332498,0.0002959969,0.0001730235,0.005048403],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006625595,"threshold_uncertainty_score":0.01317406,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W217970951","doi":"","title":"Roles of Pre-Training and Fine-Tuning in Context-Dependent DBN-HMMs for Real-World Speech Recognition","year":2010,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":193,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Hidden Markov model; Computer science; Artificial intelligence; Speech recognition; Context (archaeology); Deep belief network; Vocabulary; Deep learning; Pattern recognition (psychology)","authors":[{"name":"Dong Yu","is_ca":false},{"name":"Li Deng","is_ca":false},{"name":"George E. Dahl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0420777987342523,"gpt":0.2835390570334626,"spread":0.2414612582992103,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002368216,0.0009037516,0.0006792231,0.0002482289,0.0004271232,0.0008939229,0.001014967,0.0009694697,0.00114991],"category_scores_gemma":[0.01076601,0.0005981241,0.0002994373,0.0002687143,0.0007572976,0.001834701,0.0008233417,0.002285954,0.0004435473],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005206584,"about_ca_system_score_gemma":0.0007364769,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002723801,"about_ca_topic_score_gemma":0.005689869,"domain_scores_codex":[0.998961,0.0004882344,0.00007848669,0.0002401548,0.000145004,0.00008711986],"domain_scores_gemma":[0.9953188,0.003090937,0.0001984696,0.0008934874,0.0003424922,0.0001556473],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001019309,0.000948187,0.00597988,0.0002653667,0.0001215214,0.0001828043,0.0003519286,0.3391735,0.1617721,0.004298137,0.001051577,0.4848358],"study_design_scores_gemma":[0.0000393183,0.0002396897,0.003631412,0.00003471152,0.00003743851,0.000111494,0.00005617072,0.900795,0.09020231,0.003559868,0.001249311,0.00004330766],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2326293,0.001007637,0.7605448,0.000394664,0.0001078977,0.0001296132,0.00007674225,0.002095452,0.003013918],"genre_scores_gemma":[0.8010703,0.0002353623,0.1970919,0.0001770892,0.00002824368,0.0001128555,0.0001499794,0.0001505796,0.0009836755],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002723801,"threshold_uncertainty_score":0.01252443,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2076794394","doi":"10.1109/asru.2011.6163900","title":"Making Deep Belief Networks effective for large vocabulary continuous speech recognition","year":2011,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":192,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Speech recognition; Vocabulary; Artificial intelligence; Natural language processing; Linguistics","authors":[{"name":"Tara N. Sainath","is_ca":false},{"name":"Brian Kingsbury","is_ca":false},{"name":"Bhuvana Ramabhadran","is_ca":false},{"name":"Petr Fousek","is_ca":false},{"name":"Petr Novák","is_ca":false},{"name":"Abdelrahman Mohamed","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04830958816975836,"gpt":0.2643455917612188,"spread":0.2160360035914604,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001297158,0.0007358736,0.0005477016,0.0004273293,0.0003299831,0.001043908,0.001035911,0.0009514006,0.004128498],"category_scores_gemma":[0.006078931,0.0006556108,0.0003519813,0.0003917302,0.0007074067,0.002914418,0.001375687,0.002231536,0.001189928],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008076259,"about_ca_system_score_gemma":0.0007873181,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005079719,"about_ca_topic_score_gemma":0.007291724,"domain_scores_codex":[0.9994627,0.0001660913,0.00002638123,0.0001261462,0.0001407238,0.00007811171],"domain_scores_gemma":[0.9983324,0.001037031,0.00008503431,0.0001758301,0.0003077004,0.0000619816],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004390316,0.0002063418,0.0009147355,0.0001470849,0.00007195662,0.0001106881,0.0001310766,0.5736156,0.01493659,0.02566428,0.005006709,0.378756],"study_design_scores_gemma":[0.00001379068,0.00001905399,0.0000680776,0.000005927013,0.000006261285,0.000006334335,0.00001092543,0.9884229,0.002240978,0.008684427,0.0005173887,0.000003806087],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0317594,0.0004652266,0.9616513,0.000697438,0.00009693654,0.00003924142,0.0001091309,0.001973194,0.00320817],"genre_scores_gemma":[0.7083652,0.0004673005,0.2849285,0.0002558751,0.00009066411,0.0001237407,0.0002994565,0.0002202832,0.005249013],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005079719,"threshold_uncertainty_score":0.01381117,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2036242736","doi":"10.1109/icassp.2013.6638952","title":"A deep convolutional neural network using heterogeneous pooling for trading acoustic invariance with phonetic confusion","year":2013,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":188,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"","keywords":"Computer science; Word error rate; Speech recognition; Pooling; Convolutional neural network; TIMIT; Artificial intelligence; Dropout (neural networks); Deep learning; Artificial neural network; Hidden Markov model; Pattern recognition (psychology); Machine learning","authors":[{"name":"Li Deng","is_ca":false},{"name":"Ossama Abdel‐Hamid","is_ca":true},{"name":"Dong Yu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03207903847495538,"gpt":0.2313530530693191,"spread":0.1992740145943637,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005888924,0.0009642357,0.0005129816,0.0003937139,0.0003447289,0.000567027,0.001341429,0.0007245412,0.002002529],"category_scores_gemma":[0.001025465,0.0004128813,0.0004938471,0.0005160319,0.0004572139,0.001221158,0.001066063,0.0008218795,0.0006409706],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009042357,"about_ca_system_score_gemma":0.001027798,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006980081,"about_ca_topic_score_gemma":0.01408261,"domain_scores_codex":[0.9996825,0.00003155014,0.00001506482,0.0001045212,0.0001047176,0.0000616757],"domain_scores_gemma":[0.9997932,0.00005366368,0.00002400506,0.00004902461,0.00005773256,0.00002249186],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003174313,0.0002171709,0.002570908,0.0001549322,0.0002358794,0.0002906526,0.00008469886,0.3076307,0.1595135,0.01192551,0.006428211,0.5106303],"study_design_scores_gemma":[0.00001398822,0.00006746922,0.0005862334,0.000009980908,0.00003962398,0.00008659593,0.000005395121,0.9684488,0.02579026,0.002435148,0.002500912,0.00001561505],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03272922,0.0003933572,0.9619632,0.0001439036,0.00008299365,0.00005018734,0.0001883085,0.001829135,0.002619695],"genre_scores_gemma":[0.5793263,0.0003583436,0.4114891,0.0002608399,0.0000591615,0.0001364895,0.0008739488,0.0002111369,0.007284734],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006980081,"threshold_uncertainty_score":0.01387894,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2125964738","doi":"10.1109/icassp.2011.5947401","title":"Large vocabulary continuous speech recognition with context-dependent DBN-HMMS","year":2011,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":182,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Hidden Markov model; Computer science; Speech recognition; Word error rate; Mixture model; Artificial intelligence; Context (archaeology); Vocabulary; Phone; Pattern recognition (psychology); Sentence","authors":[{"name":"George E. Dahl","is_ca":true},{"name":"Dong Yu","is_ca":false},{"name":"Li Deng","is_ca":false},{"name":"Alex Acero","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03873870180704939,"gpt":0.2173281653530459,"spread":0.1785894635459966,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007801307,0.0006699357,0.0006722596,0.0004268428,0.0002833501,0.0006114172,0.001048041,0.0006916414,0.002615115],"category_scores_gemma":[0.002124981,0.0004368625,0.0004784302,0.0004892323,0.0002552487,0.001208756,0.0008559671,0.001147069,0.002022993],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004030259,"about_ca_system_score_gemma":0.0005965627,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007610193,"about_ca_topic_score_gemma":0.01581708,"domain_scores_codex":[0.999471,0.0001215294,0.00003639033,0.0002041923,0.0001263782,0.0000405554],"domain_scores_gemma":[0.9992557,0.0003143018,0.00004188246,0.0001583627,0.0001925434,0.00003716353],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0005226141,0.0002507682,0.001945298,0.000234108,0.0001980568,0.0001886865,0.0001529294,0.09592252,0.09592044,0.002908314,0.007149784,0.7946066],"study_design_scores_gemma":[0.00002218716,0.00006027439,0.001414092,0.00001469612,0.00003764695,0.00009687489,0.00003036327,0.9710551,0.0231626,0.001981627,0.002100313,0.00002416183],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05422433,0.001421733,0.931773,0.0001949526,0.0001910725,0.00006446795,0.0009603416,0.008815613,0.002354379],"genre_scores_gemma":[0.5825779,0.0006567136,0.4079395,0.0002480755,0.0001004669,0.0001695161,0.003026702,0.0003142045,0.004967069],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007610193,"threshold_uncertainty_score":0.01513183,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2989571531","doi":"10.1109/sped.2019.8906599","title":"FoR: A Dataset for Synthetic Speech Detection","year":2019,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":169,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"","keywords":"Computer science; Speech recognition; Artificial intelligence; Natural language processing","authors":[{"name":"Ricardo Reimao","is_ca":true},{"name":"Vassilios Tzerpos","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02481780576436919,"gpt":0.2639481081266526,"spread":0.2391303023622834,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001167251,0.004192942,0.001607587,0.002997367,0.001128402,0.001332852,0.002877084,0.002996895,0.01633849],"category_scores_gemma":[0.003806616,0.0004479745,0.001515714,0.002184469,0.0005430497,0.001009581,0.001789241,0.002019739,0.026404],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001189809,"about_ca_system_score_gemma":0.001484734,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01315169,"about_ca_topic_score_gemma":0.02850116,"domain_scores_codex":[0.9981187,0.0004025042,0.0002660517,0.0004659971,0.0005553759,0.0001912731],"domain_scores_gemma":[0.9977046,0.0006202626,0.0001759503,0.0005195807,0.0007364329,0.0002432719],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001327638,0.0009408128,0.005155356,0.002096428,0.0003083843,0.0008773571,0.0002415272,0.005727177,0.009279026,0.0009837461,0.8775744,0.09548806],"study_design_scores_gemma":[0.0009866751,0.0009401666,0.02923094,0.0005241001,0.0002824902,0.003306888,0.0009461319,0.04402836,0.02023329,0.002658878,0.8964717,0.0003905652],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.0237493,0.001736927,0.006554581,0.0005123426,0.0008133026,0.0006060297,0.9511297,0.00899014,0.005907728],"genre_scores_gemma":[0.01207643,0.0002073721,0.007330823,0.0001414158,0.00007581643,0.0005404853,0.976612,0.000207731,0.002807929],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.01633849,"threshold_uncertainty_score":0.0546577,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1749856071","doi":"","title":"A new algorithm for the alignment of phonetic sequences","year":2000,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":167,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Similarity (geometry); Algorithm; Basis (linear algebra); Scheme (mathematics); Sequence (biology); Phonology; Multiple sequence alignment; Artificial intelligence; Speech recognition; Sequence alignment; Mathematics; Image (mathematics); Linguistics","authors":[{"name":"Grzegorz Kondrak","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02299981020833271,"gpt":0.2477532114839289,"spread":0.2247534012755962,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001980484,0.001824988,0.001742714,0.003378571,0.002257758,0.003623916,0.003310404,0.002918188,0.01771958],"category_scores_gemma":[0.008012691,0.001265354,0.001788333,0.003507993,0.001820223,0.005472222,0.003389508,0.004021844,0.01330787],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008670742,"about_ca_system_score_gemma":0.001690784,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002082611,"about_ca_topic_score_gemma":0.002308062,"domain_scores_codex":[0.9969274,0.0004302298,0.0003221332,0.001103385,0.001054449,0.0001623173],"domain_scores_gemma":[0.9977998,0.0007683873,0.0001558506,0.0004088603,0.0007769674,0.00009012133],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001563535,0.00006999102,0.0004955342,0.0002608716,0.00008365116,0.000155089,0.0003118543,0.02287108,0.0118956,0.04599721,0.01549909,0.9022036],"study_design_scores_gemma":[0.000225654,0.0003390696,0.0008583948,0.0001936848,0.0001251991,0.001466868,0.0002874425,0.5963488,0.02140625,0.2018508,0.17672,0.000177853],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0005975416,0.0001287593,0.9967561,0.0000668437,0.0001517131,0.00007015048,0.00008042975,0.001190014,0.000958378],"genre_scores_gemma":[0.005658611,0.0001352158,0.9907495,0.00008363251,0.00009293826,0.0001667782,0.0003592459,0.0003126116,0.002441441],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01771958,"threshold_uncertainty_score":0.05927783,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2123798005","doi":"10.1109/icassp.2010.5495646","title":"Multilingual acoustic modeling for speech recognition based on subspace Gaussian Mixture Models","year":2010,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":160,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"","keywords":"Subspace topology; Computer science; Phone; Gaussian; Mixture model; Speech recognition; Set (abstract data type); Language model; Artificial intelligence; Acoustic model; Space (punctuation); Hidden Markov model; Natural language processing; Pattern recognition (psychology); Speech processing; Linguistics","authors":[{"name":"Lukáš Burget","is_ca":false},{"name":"Petr Schwarz","is_ca":false},{"name":"Mohit Agarwal","is_ca":false},{"name":"Pinar Akyazi","is_ca":false},{"name":"Kai Feng","is_ca":false},{"name":"Arnab Ghoshal","is_ca":false},{"name":"Ondřej Glembek","is_ca":false},{"name":"Nagendra Kumar Goel","is_ca":false},{"name":"Martin Karafiát","is_ca":false},{"name":"Daniel Povey","is_ca":false},{"name":"Ariya Rastrow","is_ca":false},{"name":"Richard C. Rose","is_ca":true},{"name":"Samuel Thomas","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04797919133278814,"gpt":0.2764242881821557,"spread":0.2284450968493676,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000914791,0.001058392,0.001015409,0.0007152215,0.0004351114,0.0009027064,0.0008956591,0.0006008152,0.003196979],"category_scores_gemma":[0.002110029,0.0004283795,0.001158996,0.0009351349,0.00031952,0.001606859,0.0009907188,0.001581667,0.003416391],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005334787,"about_ca_system_score_gemma":0.0007269831,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006187226,"about_ca_topic_score_gemma":0.01094678,"domain_scores_codex":[0.9990221,0.0003802624,0.00004681625,0.0002339643,0.000250121,0.00006669248],"domain_scores_gemma":[0.9993916,0.0002397454,0.00003471465,0.0001367236,0.0001716514,0.00002555809],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003087628,0.0001391817,0.001099753,0.0001346612,0.0002657484,0.0001185513,0.0002243982,0.3819588,0.03483895,0.01531281,0.005258772,0.5603396],"study_design_scores_gemma":[0.000005460306,0.00002174373,0.0001933809,0.000005639951,0.00001659962,0.00004359611,0.00001648249,0.9862859,0.004917082,0.006388746,0.002088604,0.00001671117],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00520249,0.0002095788,0.9915658,0.00007213752,0.00002963297,0.00001793692,0.0001175139,0.002158677,0.0006262511],"genre_scores_gemma":[0.2237241,0.000747618,0.767534,0.0001519484,0.00008857189,0.0002476397,0.002020557,0.001034991,0.004450514],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006187226,"threshold_uncertainty_score":0.01230246,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2137743578","doi":"10.1109/34.845379","title":"Training hidden Markov models with multiple observations-a combinatorial method","year":2000,"lang":"en","type":"article","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":150,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Polytechnique Montréal; Université Laval; SNC-Lavalin (Canada)","funders":"Natural Sciences and Engineering Research Council of Canada; Guangxi University; East China Institute of Technology","keywords":"Hidden Markov model; Computer science; Artificial intelligence; Markov chain; Lagrange multiplier; Generality; Handwriting; Independence (probability theory); Machine learning; Function (biology); Maximization; Pattern recognition (psychology); Algorithm; Mathematics; Mathematical optimization; Statistics","authors":[{"name":"Xiaolin Li","is_ca":true},{"name":"Marc Parizeau","is_ca":true},{"name":"Réjean Plamondon","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06005585691777636,"gpt":0.2821406138225669,"spread":0.2220847569047905,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002553301,0.0008974107,0.001255603,0.0008685568,0.000397183,0.001058027,0.00207265,0.001226868,0.003840941],"category_scores_gemma":[0.008834982,0.001200224,0.001219009,0.001285725,0.0008733423,0.002238468,0.001704166,0.002101999,0.0008723394],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000889351,"about_ca_system_score_gemma":0.001315878,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001889636,"about_ca_topic_score_gemma":0.002806941,"domain_scores_codex":[0.9985211,0.0008071142,0.00007730305,0.0002338936,0.0002665868,0.00009393108],"domain_scores_gemma":[0.9948959,0.004256693,0.0002179353,0.0002743042,0.0002806463,0.00007452447],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004020291,0.00004973567,0.0003854299,0.0001000761,0.00005290594,0.0000826396,0.00006786096,0.8498776,0.0008439886,0.06863876,0.0008896872,0.07897114],"study_design_scores_gemma":[0.000006152279,0.00001004351,0.000035633,0.000008704389,0.000007766403,0.00001259574,0.000003488957,0.9812428,0.0002452654,0.01790267,0.0005191183,0.000005743021],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001125684,0.00006791747,0.9981633,0.00004881928,0.000009501598,0.00001155492,0.00001692906,0.00009085086,0.0004654998],"genre_scores_gemma":[0.1695746,0.0005944196,0.8248267,0.0001748979,0.0001359559,0.0003662375,0.0003998993,0.0002201581,0.003707084],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003840941,"threshold_uncertainty_score":0.01350331,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2115599677","doi":"10.1109/jstsp.2010.2081790","title":"Diarization of Telephone Conversations Using Factor Analysis","year":2010,"lang":"en","type":"article","venue":"IEEE Journal of Selected Topics in Signal Processing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":141,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Computer Research Institute of Montréal","funders":"Johns Hopkins University","keywords":"Speaker diarisation; Computer science; Cluster analysis; Speech recognition; Word error rate; Speaker recognition; NIST; Channel (broadcasting); Hierarchical clustering; Bayes' theorem; Exploit; Artificial intelligence; Pattern recognition (psychology); Telecommunications; Bayesian probability","authors":[{"name":"Patrick Kenny","is_ca":true},{"name":"Douglas A. Reynolds","is_ca":false},{"name":"Fabio Castaldo","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02582700289371216,"gpt":0.2748875302374324,"spread":0.2490605273437203,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002230769,0.001309671,0.0009408662,0.001964402,0.0005973279,0.001185256,0.000859348,0.0004662008,0.00324054],"category_scores_gemma":[0.007122959,0.0003281888,0.001314765,0.001961299,0.0004325826,0.001318497,0.000767665,0.001182021,0.001855853],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006763722,"about_ca_system_score_gemma":0.0006505211,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006423189,"about_ca_topic_score_gemma":0.009112255,"domain_scores_codex":[0.9976221,0.0007538295,0.0001431182,0.0007410077,0.0005730254,0.0001669941],"domain_scores_gemma":[0.9971411,0.001178265,0.0002209346,0.000434671,0.0009438791,0.00008107131],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004726558,0.0001201785,0.003241588,0.0001899824,0.0002227011,0.00007489498,0.0006831689,0.01590695,0.04016329,0.002357534,0.002504723,0.9340624],"study_design_scores_gemma":[0.0001114365,0.0005990315,0.02400707,0.0000890391,0.0003829659,0.0007825692,0.0007822103,0.8252556,0.09609282,0.01642957,0.03513077,0.0003369859],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02795506,0.0006165972,0.9674878,0.0001077507,0.00008765247,0.0001418297,0.000342751,0.001712247,0.001548329],"genre_scores_gemma":[0.2561314,0.0007354892,0.7362862,0.0000881409,0.000149191,0.0001575858,0.001730431,0.0004989171,0.004222651],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006423189,"threshold_uncertainty_score":0.01277155,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2112021726","doi":"10.1109/taslp.2014.2346313","title":"Fast Adaptation of Deep Neural Network Based on Discriminant Codes for Speech Recognition","year":2014,"lang":"en","type":"article","venue":"IEEE/ACM Transactions on Audio Speech and Language Processing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":139,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"University of Science and Technology of China","keywords":"Computer science; Speech recognition; TIMIT; Normalization (sociology); Adaptation (eye); Artificial neural network; Artificial intelligence; Speaker recognition; Pattern recognition (psychology); Word error rate; Hidden Markov model","authors":[{"name":"Shaofei Xue","is_ca":false},{"name":"Ossama Abdel‐Hamid","is_ca":true},{"name":"Hui Jiang","is_ca":true},{"name":"Li-Rong Dai","is_ca":false},{"name":"Qingfeng Liu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02681299895984912,"gpt":0.2610017309285276,"spread":0.2341887319686785,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005802385,0.0006734337,0.000388164,0.0003923998,0.0001627367,0.0002543833,0.0005512417,0.0004450741,0.001259639],"category_scores_gemma":[0.001344353,0.0002552411,0.0004215731,0.0004057263,0.0003435254,0.0007422698,0.0006124825,0.001353437,0.0005576932],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004679637,"about_ca_system_score_gemma":0.0004740162,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003717159,"about_ca_topic_score_gemma":0.006174526,"domain_scores_codex":[0.9996969,0.00006472963,0.0000154087,0.00008597549,0.000105456,0.00003157762],"domain_scores_gemma":[0.9996141,0.0001363855,0.00002982519,0.00006951269,0.0001317024,0.00001864518],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002554265,0.0001247972,0.001983158,0.0001236891,0.00009517022,0.0001094503,0.0001091943,0.2647353,0.145533,0.004231662,0.003361296,0.5793378],"study_design_scores_gemma":[0.000006799437,0.00003743316,0.0008714966,0.000006986725,0.00001300929,0.00004494863,0.00001068086,0.9654099,0.03027317,0.001412329,0.001897275,0.00001593266],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04423909,0.0007818028,0.9513497,0.0001136308,0.0001382448,0.00004451408,0.00009790861,0.001806843,0.001428282],"genre_scores_gemma":[0.6446325,0.000726824,0.3482639,0.0001900712,0.00006517684,0.0001404726,0.0005820689,0.0002318648,0.005167224],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003717159,"threshold_uncertainty_score":0.007391036,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2398971481","doi":"10.21437/interspeech.2015-95","title":"The reddots data collection for speaker recognition","year":2015,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":138,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"VoiceAge (Canada); Computer Research Institute of Montréal","funders":"","keywords":"Computer science; Speech recognition; Speaker recognition; Data collection; Natural language processing; Artificial intelligence; Mathematics","authors":[{"name":"Kong Aik Lee","is_ca":false},{"name":"Anthony Larcher","is_ca":false},{"name":"Guangsen Wang","is_ca":false},{"name":"Patrick Kenny","is_ca":true},{"name":"Niko Brümmer","is_ca":false},{"name":"David A. van Leeuwen","is_ca":false},{"name":"Hagai Aronowitz","is_ca":false},{"name":"Marcel Kockmann","is_ca":true},{"name":"Carlos Vaquero","is_ca":false},{"name":"Bin Ma","is_ca":false},{"name":"Haizhou Li","is_ca":false},{"name":"Themos Stafylakis","is_ca":true},{"name":"Jahangir Alam","is_ca":true},{"name":"Albert Swart","is_ca":false},{"name":"Javier Pérez","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.285855109260436,"gpt":0.3252087216737488,"spread":0.03935361241331276,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003203561,0.001731844,0.001866115,0.002629422,0.001134085,0.001231506,0.0023585,0.001496231,0.03958827],"category_scores_gemma":[0.007696688,0.0005482924,0.00109243,0.001663022,0.0006815023,0.001096944,0.002748525,0.001565052,0.07792414],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006650438,"about_ca_system_score_gemma":0.002305301,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01604926,"about_ca_topic_score_gemma":0.02886944,"domain_scores_codex":[0.9957322,0.001046674,0.0004490991,0.0008524529,0.001582835,0.0003367355],"domain_scores_gemma":[0.9905269,0.0008834667,0.0004415204,0.00279091,0.004619694,0.0007375062],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001744039,0.000500329,0.00875964,0.0009703119,0.0001868055,0.0002893141,0.0003286894,0.001871039,0.01203619,0.001332481,0.8495368,0.1224443],"study_design_scores_gemma":[0.001211142,0.0008107804,0.08757093,0.000404335,0.0002160193,0.001187974,0.0007561301,0.01063889,0.02265983,0.002307838,0.8719327,0.0003033942],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.02867161,0.0005140876,0.02578625,0.0004700949,0.001656698,0.002105441,0.9023715,0.01293536,0.02548889],"genre_scores_gemma":[0.02169181,0.0001338521,0.01710176,0.0001582012,0.0002375864,0.00266003,0.9405276,0.00106286,0.01642641],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.03958827,"threshold_uncertainty_score":0.132436,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2008066109","doi":"10.1016/j.specom.2009.04.006","title":"Tools and Technologies for Computer-Aided Speech and Language Therapy","year":2009,"lang":"en","type":"article","venue":"Speech Communication","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":133,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Pronunciation; Speech recognition; Population; Dysarthria; Set (abstract data type); Speech technology; Articulation (sociology); Domain (mathematical analysis); Speech processing; Speech corpus; Natural language processing; Speech synthesis; Linguistics; Audiology; Medicine","authors":[{"name":"Óscar Saz","is_ca":false},{"name":"Shou-Chun Yin","is_ca":true},{"name":"Eduardo Lleida","is_ca":false},{"name":"Richard C. Rose","is_ca":true},{"name":"Carlos Vaquero","is_ca":false},{"name":"William Ricardo Rodríguez Dueñas","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03613621209312245,"gpt":0.287530217985311,"spread":0.2513940058921886,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009782535,0.0006074231,0.0004801459,0.001370468,0.0003953548,0.001858198,0.0008355408,0.0009736389,0.02341073],"category_scores_gemma":[0.002108076,0.0002506213,0.0003735719,0.000736122,0.0006289083,0.001783401,0.001614422,0.0007441778,0.006454371],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002983957,"about_ca_system_score_gemma":0.0007312082,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005809189,"about_ca_topic_score_gemma":0.0008541734,"domain_scores_codex":[0.9992954,0.0001648197,0.00005784988,0.00006253238,0.0003794279,0.00003991217],"domain_scores_gemma":[0.999128,0.0004562007,0.00004260218,0.0001378193,0.0001880647,0.00004720401],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001225677,0.00007527263,0.0003422521,0.0004883637,0.00002714416,0.0002826976,0.0003421056,0.001239216,0.03165745,0.04152733,0.01785597,0.9060395],"study_design_scores_gemma":[0.0001259959,0.0005369078,0.003076134,0.001149836,0.0001704774,0.004823703,0.0006327901,0.02068212,0.0695498,0.09418691,0.8049407,0.0001245407],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005773862,0.008962193,0.9373915,0.0008067717,0.0005706322,0.000231004,0.0003532348,0.005323655,0.04058713],"genre_scores_gemma":[0.08684479,0.01129199,0.8464442,0.0006710762,0.0003145253,0.0008001901,0.0007326561,0.0006639512,0.05223655],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02341073,"threshold_uncertainty_score":0.07831675,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2109886035","doi":"10.1109/icassp.2016.7472618","title":"End-to-end attention-based large vocabulary speech recognition","year":2016,"lang":"en","type":"preprint","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":124,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal","funders":"Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Computer science; Hidden Markov model; Recurrent neural network; Speech recognition; Decoding methods; Vocabulary; Sequence (biology); Pooling; Artificial intelligence; Character (mathematics); Language model; Process (computing); Pattern recognition (psychology); Artificial neural network; Algorithm","authors":[{"name":"Dzmitry Bahdanau","is_ca":true},{"name":"Jan Chorowski","is_ca":false},{"name":"Dmitriy Serdyuk","is_ca":true},{"name":"Philémon Brakel","is_ca":true},{"name":"Yoshua Bengio","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03520127758503985,"gpt":0.2663043535830252,"spread":0.2311030759979854,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000829407,0.001477215,0.00123857,0.0008922725,0.0004618996,0.001043357,0.001935661,0.001302293,0.01365783],"category_scores_gemma":[0.002630598,0.000377996,0.0005585971,0.0006390081,0.000416011,0.001441003,0.001575673,0.0013603,0.01601574],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005838146,"about_ca_system_score_gemma":0.0008332974,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004164009,"about_ca_topic_score_gemma":0.008928131,"domain_scores_codex":[0.998989,0.0001690589,0.00004720689,0.0003718975,0.000284522,0.0001382888],"domain_scores_gemma":[0.9986934,0.0004593465,0.00005155659,0.0002810703,0.0004514741,0.00006322409],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005154727,0.0002108722,0.0006257042,0.0002039246,0.00007738172,0.0002172624,0.0001029156,0.0148411,0.1017447,0.00246369,0.01588943,0.8631076],"study_design_scores_gemma":[0.00005152865,0.0001883794,0.002096532,0.00003157803,0.00006381438,0.0003936842,0.00007372964,0.8004943,0.1766763,0.008429196,0.011451,0.00004997787],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02080865,0.0006137789,0.9488205,0.0001565062,0.0002592391,0.0001741776,0.001342858,0.0211108,0.006713575],"genre_scores_gemma":[0.3430458,0.0004651333,0.6216968,0.000434281,0.0002400266,0.0003578555,0.008018047,0.00115934,0.02458281],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01365783,"threshold_uncertainty_score":0.04569,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4291746033","doi":"10.3390/app12094419","title":"Automatic Speech Recognition (ASR) Systems for Children: A Systematic Literature Review","year":2022,"lang":"en","type":"article","venue":"Applied Sciences","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":124,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Moncton","funders":"","keywords":"Speech recognition; Computer science; Task (project management); Speech technology; Field (mathematics); Process (computing); Speech processing; Engineering","authors":[{"name":"Vivek Bhardwaj","is_ca":false},{"name":"Mohamed Tahar Ben Othman","is_ca":false},{"name":"Vinay Kukreja","is_ca":false},{"name":"Youcef Belkhier","is_ca":false},{"name":"Mohit Bajaj","is_ca":false},{"name":"B. Srikanth Goud","is_ca":false},{"name":"Ateeq Ur Rehman","is_ca":true},{"name":"Muhammad Shafiq","is_ca":false},{"name":"Habib Hamam","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02825174921800579,"gpt":0.2559027336825049,"spread":0.2276509844644991,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005884323,0.001355398,0.004006503,0.009729762,0.0005657581,0.001941878,0.00199332,0.001731825,0.005306277],"category_scores_gemma":[0.02390003,0.0007961899,0.004379223,0.007480117,0.0009545198,0.002803159,0.001180388,0.001030851,0.0008585675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002118867,"about_ca_system_score_gemma":0.01020455,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008622719,"about_ca_topic_score_gemma":0.0194846,"domain_scores_codex":[0.9951311,0.00114846,0.001934516,0.00051548,0.001147944,0.0001224858],"domain_scores_gemma":[0.9732527,0.02111777,0.002462045,0.0002965673,0.002707454,0.0001634941],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0001090821,0.00003644578,0.000850371,0.6787756,0.001390012,0.0001452897,0.0002608795,0.0001781699,0.0002644637,0.0004967311,0.003855156,0.3136378],"study_design_scores_gemma":[0.0001038594,0.0003623531,0.006529241,0.8805122,0.01746177,0.001204843,0.0006398477,0.0002165868,0.0005401264,0.0006785067,0.09168706,0.00006357586],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0003215143,0.998912,0.0001540362,0.0001589489,0.00004709731,0.00006193731,0.0001233689,0.000005864416,0.0002152472],"genre_scores_gemma":[0.002642699,0.9962263,0.0006610153,0.000162376,0.00003207506,0.0001110202,0.0001105398,0.000002976744,0.00005104828],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.009729762,"threshold_uncertainty_score":0.03111964,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2015633636","doi":"10.1109/icassp.2014.6854823","title":"I-vector-based speaker adaptation of deep neural networks for French broadcast audio transcription","year":2014,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":119,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Computer Research Institute of Montréal","funders":"","keywords":"Speech recognition; Computer science; Speaker diarisation; Word error rate; Transcription (linguistics); Feature vector; Hidden Markov model; Artificial neural network; Artificial intelligence; Speaker recognition; Acoustic model; Vector quantization; Pattern recognition (psychology); Speech processing","authors":[{"name":"Vishwa Gupta","is_ca":true},{"name":"Patrick Kenny","is_ca":true},{"name":"Pierre Ouellet","is_ca":true},{"name":"Themos Stafylakis","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02772913931601576,"gpt":0.2293340230087177,"spread":0.2016048836927019,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006859531,0.0006102549,0.0002904318,0.0002654243,0.0001817251,0.0002882918,0.0005072827,0.0004317035,0.003215957],"category_scores_gemma":[0.001431727,0.0002271391,0.0003660003,0.0002492857,0.0001931133,0.0004497302,0.0005068204,0.0009486963,0.001410047],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004378377,"about_ca_system_score_gemma":0.000353314,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00644605,"about_ca_topic_score_gemma":0.00905784,"domain_scores_codex":[0.9996284,0.0001391691,0.00001892623,0.00009540556,0.00007044574,0.00004757124],"domain_scores_gemma":[0.9996321,0.0001582268,0.00002098258,0.0000477787,0.0001226398,0.00001829473],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000586083,0.0001354138,0.001106579,0.0001143654,0.0001099996,0.0001340279,0.00020678,0.1451586,0.1525994,0.001075069,0.004083545,0.6946902],"study_design_scores_gemma":[0.0000214259,0.0001374066,0.002034944,0.00001377332,0.00004035027,0.00009806074,0.00004934699,0.9198104,0.07425089,0.0008726233,0.002638055,0.00003269261],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1585557,0.0009357904,0.8246348,0.0002273546,0.0002302033,0.0001033799,0.0005158968,0.009869349,0.004927555],"genre_scores_gemma":[0.7556191,0.0003755632,0.2311812,0.0001723339,0.00007878579,0.0001710985,0.001761668,0.0005784988,0.0100618],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00644605,"threshold_uncertainty_score":0.01281708,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1846073453","doi":"10.1109/icassp.2000.859138","title":"Towards language independent acoustic modeling","year":2002,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":111,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"Universidade Federal de Alagoas; Rice University","keywords":"Computer science; Czech; Discriminative model; Hidden Markov model; Acoustic model; Language model; Speech recognition; Natural language processing; Mandarin Chinese; Artificial intelligence; Adaptation (eye); Speech processing; Linguistics","authors":[{"name":"Bill Byrne","is_ca":false},{"name":"Peter Beyerlein","is_ca":false},{"name":"Juan M. Huerta","is_ca":false},{"name":"Sanjeev Khudanpur","is_ca":false},{"name":"Bhaskara Marthi","is_ca":true},{"name":"John Morgan","is_ca":false},{"name":"Nino Peterek","is_ca":false},{"name":"J. Picone","is_ca":false},{"name":"Dimitra Vergyri","is_ca":false},{"name":"T. Wang","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04217330464563866,"gpt":0.2469683376392459,"spread":0.2047950329936072,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001396284,0.001468266,0.0009631688,0.0007825848,0.0006358511,0.001707457,0.001683839,0.001077069,0.004141639],"category_scores_gemma":[0.00306411,0.0008909735,0.002015291,0.0006857302,0.000625072,0.00185815,0.001876485,0.002521119,0.007714354],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004311954,"about_ca_system_score_gemma":0.001113963,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002677912,"about_ca_topic_score_gemma":0.003970749,"domain_scores_codex":[0.9983804,0.0004442146,0.00006617096,0.0004698818,0.0005550459,0.00008423946],"domain_scores_gemma":[0.9985635,0.0003805624,0.00006563612,0.0004770494,0.0004679139,0.00004542013],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002894416,0.000212259,0.001679265,0.0003737723,0.0003487716,0.000305587,0.0006219222,0.2719954,0.1503814,0.04421491,0.008081771,0.5214955],"study_design_scores_gemma":[0.00003118285,0.00008248757,0.0006633384,0.00004107173,0.00008672079,0.0002442949,0.0001021051,0.9087569,0.03556014,0.02424183,0.0301094,0.00008061741],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002708348,0.0001156708,0.9927127,0.00009369066,0.00006523867,0.00002937566,0.0001579183,0.001697953,0.002419225],"genre_scores_gemma":[0.1072553,0.0006510031,0.8783106,0.0003550128,0.0001377537,0.0003045662,0.002597298,0.001452036,0.008936524],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004141639,"threshold_uncertainty_score":0.0138551,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2167191127","doi":"10.1109/icassp.1989.266415","title":"A comparison of several acoustic representations for speech recognition with degraded and undegraded speech","year":2003,"lang":"en","type":"article","venue":"International Conference on Acoustics, Speech, and Signal Processing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":109,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"National Research Council Canada","funders":"","keywords":"Speech recognition; Cepstrum; Mel-frequency cepstrum; Computer science; Weighting; Noise (video); Filter (signal processing); Filter bank; Pattern recognition (psychology); Mathematics; Artificial intelligence; Feature extraction; Acoustics","authors":[{"name":"Melvyn J. Hunt","is_ca":true},{"name":"Corentin Lefebvre","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1034577501804792,"gpt":0.3427565667719608,"spread":0.2392988165914816,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002201583,0.0008172153,0.0004942533,0.001109848,0.0001931528,0.001016109,0.0006730722,0.0006541306,0.001767889],"category_scores_gemma":[0.007881362,0.0002132309,0.0006860453,0.0005290668,0.0003230202,0.001424047,0.0006130476,0.000515302,0.00104198],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004393665,"about_ca_system_score_gemma":0.0004080252,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001981316,"about_ca_topic_score_gemma":0.001968136,"domain_scores_codex":[0.9990731,0.0002848521,0.00008630734,0.0001910784,0.0002924996,0.00007206361],"domain_scores_gemma":[0.9974393,0.001528392,0.0001323999,0.0002661312,0.0005353014,0.00009841709],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.006472432,0.0005604306,0.01049801,0.0004724853,0.0003960998,0.0001633455,0.0003994448,0.05550773,0.1062764,0.001610441,0.001111361,0.8165318],"study_design_scores_gemma":[0.0002943798,0.004867421,0.044947,0.0001261336,0.0005826003,0.00124222,0.0005231497,0.8051161,0.1370951,0.002018159,0.002901046,0.0002867874],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6708349,0.0008492456,0.3214447,0.0001721526,0.0001132023,0.0002219567,0.000757509,0.002517442,0.003088951],"genre_scores_gemma":[0.8897381,0.0003227423,0.1068307,0.00004716938,0.00001884433,0.0001308001,0.001248876,0.000105143,0.001557409],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002201583,"threshold_uncertainty_score":0.01164323,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3210530853","doi":"10.1109/icassp43922.2022.9746484","title":"A Comparison of Discrete and Soft Speech Units for Improved Voice Conversion","year":2022,"lang":"en","type":"article","venue":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":107,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Ubisoft (Canada)","funders":"","keywords":"Computer science; Speech recognition","authors":[{"name":"Benjamin van Niekerk","is_ca":true},{"name":"Marc‐André Carbonneau","is_ca":true},{"name":"Julian Zaïdi","is_ca":true},{"name":"Matthew Baas","is_ca":false},{"name":"Hugo Seuté","is_ca":true},{"name":"Herman Kamper","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07575626700609553,"gpt":0.3305673919311458,"spread":0.2548111249250503,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001209038,0.0004504875,0.000471233,0.0005503509,0.0001629992,0.0009296467,0.0005874094,0.0005066113,0.003431899],"category_scores_gemma":[0.003803303,0.0001418551,0.0003926553,0.0003563632,0.0004863539,0.00114959,0.0008019529,0.0008565289,0.0007035811],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002974666,"about_ca_system_score_gemma":0.0003003388,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005619076,"about_ca_topic_score_gemma":0.0006469668,"domain_scores_codex":[0.9993349,0.000222741,0.0000509356,0.0001270294,0.000207047,0.00005732515],"domain_scores_gemma":[0.9986725,0.0007691892,0.00006103824,0.0002040906,0.0002080747,0.00008510458],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002142006,0.0004010283,0.001475763,0.0002943662,0.00009428184,0.0001408914,0.0001400475,0.1048531,0.08213743,0.00556331,0.0009909287,0.8017669],"study_design_scores_gemma":[0.0001064038,0.001197722,0.003214845,0.00003844069,0.00007135707,0.0001981574,0.000116817,0.9131969,0.07586293,0.003846366,0.002107978,0.00004210866],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4926011,0.002761368,0.4896672,0.000439414,0.0003062563,0.0001305278,0.0001777188,0.002763826,0.01115261],"genre_scores_gemma":[0.9221697,0.0002942843,0.07466713,0.0001168072,0.00004151417,0.00005273872,0.0002050524,0.0001183467,0.002334418],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003431899,"threshold_uncertainty_score":0.01148081,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2151936436","doi":"10.1109/tasl.2010.2072499","title":"Articulatory Knowledge in the Recognition of Dysarthric Speech","year":2010,"lang":"en","type":"article","venue":"IEEE Transactions on Audio Speech and Language Processing","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":105,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Vocal tract; Discriminative model; Dysarthria; Speech recognition; Computer science; Speech production; Manner of articulation; Generative grammar; Artificial intelligence; Natural language processing; Psychology","authors":[{"name":"Frank Rudzicz","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01747474294777467,"gpt":0.2601078714734097,"spread":0.242633128525635,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001353385,0.0003930754,0.0003461106,0.0006192945,0.0002832183,0.0008444682,0.0005168959,0.0005674831,0.0007896985],"category_scores_gemma":[0.005975587,0.0001929492,0.0002067538,0.0003501641,0.0005143932,0.001180177,0.0005633093,0.0005282744,0.000546032],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002863988,"about_ca_system_score_gemma":0.000463,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003728272,"about_ca_topic_score_gemma":0.0070201,"domain_scores_codex":[0.9994499,0.0001752316,0.00003543099,0.0001669349,0.0001347496,0.00003777469],"domain_scores_gemma":[0.9982802,0.001171418,0.0001395793,0.0002107309,0.0001570849,0.00004103181],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0006114172,0.0001460493,0.01362826,0.0002262543,0.00008011524,0.0003205568,0.0003387615,0.09206422,0.06257381,0.001552396,0.0005102729,0.827948],"study_design_scores_gemma":[0.00002153705,0.0003295008,0.04262976,0.0000693729,0.00008700664,0.001141672,0.0003293359,0.908073,0.0396208,0.006239749,0.001400118,0.00005813063],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8443563,0.00175606,0.1500953,0.0002324482,0.00003659681,0.00002708584,0.0001603892,0.0007009214,0.002634894],"genre_scores_gemma":[0.9870064,0.0003194605,0.01168769,0.00002493509,0.00001421535,0.000006481676,0.0001483325,0.00001667897,0.0007756826],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003728272,"threshold_uncertainty_score":0.007413149,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3206189675","doi":"10.1109/icassp43922.2022.9747814","title":"Large-Scale Self-Supervised Speech Representation Learning for Automatic Speaker Verification","year":2022,"lang":"en","type":"article","venue":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":102,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Generalizability theory; Computer science; Speech recognition; Word error rate; Speaker recognition; Artificial intelligence; Representation (politics); Artificial neural network; Scale (ratio); Feature (linguistics); Pattern recognition (psychology); Machine learning; Mathematics","authors":[{"name":"Zhengyang Chen","is_ca":true},{"name":"Sanyuan Chen","is_ca":false},{"name":"Yu Wu","is_ca":false},{"name":"Yao Qian","is_ca":true},{"name":"Chengyi Wang","is_ca":false},{"name":"Shujie Liu","is_ca":false},{"name":"Yanmin Qian","is_ca":true},{"name":"Michael Zeng","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0517667352431102,"gpt":0.3075118367600685,"spread":0.2557451015169583,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002093885,0.0009885173,0.0008306139,0.0004783709,0.0003475823,0.0005541008,0.001361725,0.0009018201,0.001583271],"category_scores_gemma":[0.003927257,0.0002922473,0.0006037761,0.0003806614,0.0004801858,0.0016079,0.001303852,0.001341132,0.001739394],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004384193,"about_ca_system_score_gemma":0.0007770734,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002283589,"about_ca_topic_score_gemma":0.003038761,"domain_scores_codex":[0.9984571,0.0005954118,0.0000569385,0.0004752613,0.000318901,0.00009643196],"domain_scores_gemma":[0.998173,0.0005595586,0.0001298114,0.0006718581,0.0004096714,0.00005604734],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004968165,0.0003324261,0.00155584,0.0001338944,0.0001383094,0.00006363054,0.0001167907,0.09967714,0.04211066,0.001947413,0.008189127,0.845238],"study_design_scores_gemma":[0.00001536351,0.0001166225,0.0008712773,0.000008662662,0.0000237253,0.00006243444,0.00002491709,0.9729192,0.02320033,0.001528821,0.001212984,0.00001573835],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1059898,0.001567086,0.8789302,0.0002610617,0.0002010005,0.0001432184,0.0005612142,0.00980564,0.002540854],"genre_scores_gemma":[0.8104081,0.0003546148,0.1809795,0.0002058219,0.0001192573,0.0002120805,0.003008436,0.0002903043,0.004421971],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002283589,"threshold_uncertainty_score":0.01107365,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2063689849","doi":"10.1109/iscslp.2012.6423452","title":"Investigation of deep neural networks (DNN) for large vocabulary continuous speech recognition: Why DNN surpasses GMMS in acoustic modeling","year":2012,"lang":"en","type":"article","venue":"","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":98,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"","keywords":"Computer science; Speech recognition; Artificial neural network; Vocabulary; Context (archaeology); Mel-frequency cepstrum; Word error rate; Artificial intelligence; Task (project management); Reduction (mathematics); Deep neural networks; Hidden Markov model; Logarithm; Pattern recognition (psychology); Feature extraction; Mathematics","authors":[{"name":"Jia Pan","is_ca":false},{"name":"Cong Liu","is_ca":false},{"name":"Zhiguo Wang","is_ca":false},{"name":"Yu Hu","is_ca":false},{"name":"Hui Jiang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05066265927163722,"gpt":0.2545038198180309,"spread":0.2038411605463937,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002349414,0.0006890955,0.0004016277,0.00041247,0.0002760655,0.0009836054,0.0006970619,0.0009600254,0.001521621],"category_scores_gemma":[0.003884432,0.0003598653,0.0002350865,0.0005264896,0.0006025077,0.002988889,0.000605821,0.001817395,0.0005674268],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008754957,"about_ca_system_score_gemma":0.000570618,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004489874,"about_ca_topic_score_gemma":0.008039345,"domain_scores_codex":[0.999486,0.0001356057,0.00002677247,0.0001349516,0.0001656764,0.00005094539],"domain_scores_gemma":[0.9985282,0.0007938984,0.00007211812,0.000132908,0.0004178041,0.0000549449],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005484296,0.000205932,0.007372536,0.000464872,0.0001317376,0.0002301875,0.000242542,0.09609335,0.0563303,0.06097311,0.007448039,0.769959],"study_design_scores_gemma":[0.0000380538,0.0004518059,0.003235974,0.0001263532,0.0000652071,0.0003134511,0.0001506276,0.8885039,0.04646576,0.04155431,0.01904738,0.00004715784],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2581106,0.04416769,0.6469222,0.01492321,0.0006964692,0.0001304103,0.0003490347,0.001629015,0.0330714],"genre_scores_gemma":[0.7712332,0.01098337,0.209672,0.001181934,0.0002876875,0.00005889409,0.0002547448,0.0001503411,0.006177748],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004489874,"threshold_uncertainty_score":0.01242507,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}