{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":41,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":41,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"84f0ed203413","filters":{"venue":"Language Resources and Evaluation"}},"results":[{"id":"W2037256905","doi":"10.1007/s10579-014-9271-6","title":"A qualitative comparison method for rhetorical structures: identifying different discourse structures in multilingual corpora","year":2014,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":81,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Rhetorical question; Linguistics; Computer science; Natural language processing; Annotation; Translation (biology); Artificial intelligence; Contrastive linguistics; Applied linguistics; Philosophy","authors":[{"name":"Mikel Iruskieta","is_ca":false},{"name":"Iria da Cunha","is_ca":false},{"name":"Maite Taboada","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08234689324121308,"gpt":0.4785165175012421,"spread":0.3961696242600291,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0441611,0.000992284,0.001165928,0.01316289,0.004194523,0.004793654,0.002542281,0.001259521,0.01152615],"category_scores_gemma":[0.1346006,0.0007095077,0.001127045,0.01009224,0.004052201,0.003713692,0.004871916,0.001491932,0.001516263],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005687786,"about_ca_system_score_gemma":0.007736226,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00460134,"about_ca_topic_score_gemma":0.009983829,"domain_scores_codex":[0.9454365,0.03518339,0.00460223,0.004537885,0.009237705,0.001002239],"domain_scores_gemma":[0.8347893,0.1099494,0.006716553,0.008818467,0.03808437,0.001641989],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003017055,0.00126629,0.02847886,0.01044309,0.0005959037,0.0005216937,0.1561338,0.002499183,0.05144858,0.1478333,0.018912,0.5788502],"study_design_scores_gemma":[0.00226944,0.002464467,0.07732867,0.00526562,0.001553017,0.001622315,0.2363618,0.05261757,0.135638,0.2517841,0.2320497,0.001045379],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1122379,0.0007877775,0.815582,0.001197241,0.0003182917,0.0184035,0.009043007,0.001220426,0.04120978],"genre_scores_gemma":[0.2523417,0.0002599571,0.7055449,0.0003151655,0.00004421625,0.03367151,0.00279752,0.0003893218,0.004635634],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0441611,"threshold_uncertainty_score":0.233549,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2062046281","doi":"10.1007/s10579-009-9083-2","title":"Classification of semantic relations between nominals","year":2009,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"National Research Council Canada; University of Ottawa","funders":"","keywords":"Computer science; Task (project management); Natural language processing; SemEval; Sentence; Artificial intelligence; Process (computing); Semantic similarity; Linguistics","authors":[{"name":"Roxana Gîrju","is_ca":false},{"name":"Preslav Nakov","is_ca":false},{"name":"Vivi Năstase","is_ca":false},{"name":"Stan Śzpakowicz","is_ca":true},{"name":"Peter D. Turney","is_ca":true},{"name":"Deniz Yüret","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03096635623433478,"gpt":0.3302474758262631,"spread":0.2992811195919283,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004882944,0.0005685298,0.0007210122,0.007751164,0.001641432,0.003738079,0.00140669,0.001187228,0.005933037],"category_scores_gemma":[0.02216887,0.0002190397,0.0008952202,0.003360093,0.001042472,0.006014624,0.00175972,0.001125461,0.001651054],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001989763,"about_ca_system_score_gemma":0.002217573,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006979066,"about_ca_topic_score_gemma":0.007146496,"domain_scores_codex":[0.9948654,0.00153901,0.0005659984,0.0007968813,0.001882087,0.0003507143],"domain_scores_gemma":[0.9796158,0.0120033,0.001306624,0.001716645,0.004554663,0.0008029657],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.003658844,0.001071379,0.1004248,0.001162219,0.0002319026,0.000582464,0.001950994,0.008358469,0.03143409,0.06389728,0.02012139,0.7671063],"study_design_scores_gemma":[0.0003284944,0.001027204,0.134528,0.0006215729,0.0008854677,0.001487108,0.008238938,0.5573241,0.08334555,0.1470077,0.06497576,0.0002302175],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7705426,0.002731259,0.1711091,0.001683264,0.0003960184,0.0007887511,0.01124628,0.004001488,0.03750122],"genre_scores_gemma":[0.9145765,0.0005040888,0.06927261,0.00008638743,0.00009531044,0.0001696638,0.01141195,0.0002225959,0.003660891],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007751164,"threshold_uncertainty_score":0.02582377,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3029927342","doi":"","title":"Contextualized Embeddings based Transformer Encoder for Sentence Similarity Modeling in Answer Selection Task","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"","keywords":"Transformer; Encoder; Computer science; Sentence; Artificial intelligence; Language model; Natural language processing; Feature selection; Selection (genetic algorithm); Engineering; Voltage","authors":[{"name":"Md Tahmid Rahman Laskar","is_ca":true},{"name":"Jimmy Xiangji Huang","is_ca":true},{"name":"Enamul Hoque","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05742582045155489,"gpt":0.309027963334743,"spread":0.2516021428831882,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00137681,0.000937986,0.0008692113,0.001379253,0.0004474274,0.0009522694,0.001090735,0.0009634807,0.007119871],"category_scores_gemma":[0.004363884,0.0002559669,0.000665853,0.001054224,0.0001954239,0.002926969,0.001312269,0.001495448,0.003540469],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004565052,"about_ca_system_score_gemma":0.001424394,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002935941,"about_ca_topic_score_gemma":0.004752193,"domain_scores_codex":[0.9988775,0.000384022,0.0000871991,0.0002998697,0.0002116027,0.000139867],"domain_scores_gemma":[0.9981695,0.0007021478,0.00007810594,0.0002260601,0.0007042358,0.0001200003],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001571566,0.0009806359,0.0055961,0.0004979068,0.000248783,0.0002868446,0.0003255237,0.02031626,0.05460699,0.009898605,0.04435333,0.8613174],"study_design_scores_gemma":[0.0001167806,0.0004947137,0.002502358,0.0000460032,0.0002140514,0.0003275677,0.0002466177,0.9375609,0.03966217,0.01192872,0.006850976,0.00004909856],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1639009,0.001644553,0.8047132,0.000653683,0.0006685387,0.0004170875,0.005566364,0.01665849,0.005777076],"genre_scores_gemma":[0.7925718,0.0005342071,0.1862885,0.0002380007,0.000245496,0.0003797489,0.0133371,0.0005880438,0.005817258],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007119871,"threshold_uncertainty_score":0.02381831,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2583305969","doi":"10.1007/s10579-017-9383-x","title":"RST Signalling Corpus: a corpus of signals of coherence relations","year":2017,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Treebank; Annotation; Corpus linguistics; Natural language processing; Computer science; Parsing; Coherence (philosophical gambling strategy); Artificial intelligence; Text corpus; Linguistics; British National Corpus","authors":[{"name":"Debopam Das","is_ca":true},{"name":"Maite Taboada","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03078082295855368,"gpt":0.3193694584320568,"spread":0.2885886354735031,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002075363,0.001164985,0.0008034284,0.004286251,0.001530585,0.00207248,0.001655261,0.002553018,0.03060549],"category_scores_gemma":[0.01715668,0.0006566732,0.0004862276,0.003571222,0.00130829,0.002169056,0.002101392,0.001761684,0.01435447],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001044953,"about_ca_system_score_gemma":0.002350146,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009325515,"about_ca_topic_score_gemma":0.01063891,"domain_scores_codex":[0.9972509,0.0008702228,0.0003310136,0.0004656031,0.0009038898,0.0001785056],"domain_scores_gemma":[0.9850116,0.009570358,0.0007219234,0.00213418,0.002116461,0.0004455129],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003904845,0.0009018078,0.00634534,0.008438109,0.0003241394,0.0026162,0.00806343,0.005638235,0.147798,0.04604514,0.4653592,0.3045656],"study_design_scores_gemma":[0.001108444,0.0005845534,0.0727355,0.0008517589,0.0004819446,0.004267394,0.00380409,0.01848018,0.07631247,0.01480362,0.8060129,0.0005571572],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.2560739,0.0047461,0.08869983,0.003343754,0.001355171,0.001667574,0.5526342,0.01364039,0.07783917],"genre_scores_gemma":[0.4079369,0.001622593,0.0760173,0.0007212487,0.0005373625,0.002497724,0.4724433,0.002643691,0.03557991],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03060549,"threshold_uncertainty_score":0.1023856,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3029889931","doi":"","title":"The Johns Hopkins University Bible Corpus: 1600+ Tongues for Typological Exploration","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Natural language processing; Variety (cybernetics); Representation (politics); Linguistics; Corpus linguistics; Parallel corpora; Pronoun; Artificial intelligence; Information retrieval; History; Machine translation; Philosophy","authors":[{"name":"Arya D. McCarthy","is_ca":false},{"name":"Rachel Wicks","is_ca":false},{"name":"Dylan Lewis","is_ca":false},{"name":"Aaron Mueller","is_ca":false},{"name":"Winston Wu","is_ca":false},{"name":"Oliver Adams","is_ca":false},{"name":"Garrett Nicolai","is_ca":true},{"name":"Matt Post","is_ca":false},{"name":"David Yarowsky","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04572153790965464,"gpt":0.2934511416866595,"spread":0.2477296037770049,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001044035,0.0005028843,0.0004775097,0.009435523,0.002489982,0.001889485,0.0007299999,0.0006235684,0.05307864],"category_scores_gemma":[0.006478014,0.000280877,0.0001629539,0.01003085,0.0011042,0.00118633,0.002365574,0.0008277936,0.02594782],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009811205,"about_ca_system_score_gemma":0.002650174,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01462967,"about_ca_topic_score_gemma":0.03021956,"domain_scores_codex":[0.998987,0.000280591,0.0001522211,0.000135184,0.0003599264,0.00008511085],"domain_scores_gemma":[0.9962239,0.001292575,0.0002337776,0.0006562498,0.0012055,0.0003878548],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0006354945,0.0002004904,0.0079096,0.002174381,0.00003442351,0.001020322,0.01150041,0.0004840266,0.01314717,0.01575643,0.7551908,0.1919464],"study_design_scores_gemma":[0.0001077273,0.00005769155,0.05234583,0.0005151813,0.00003638779,0.0006620885,0.005707761,0.0007425279,0.005066743,0.001955146,0.9327561,0.00004677055],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.2018405,0.00351904,0.007090948,0.001658021,0.001288684,0.00056106,0.6098959,0.001563343,0.1725826],"genre_scores_gemma":[0.2664446,0.00203197,0.01897835,0.0005498966,0.0005688617,0.001442448,0.6440079,0.00175504,0.06422094],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.05307864,"threshold_uncertainty_score":0.1775658,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3028807070","doi":"","title":"Extraction of Hyponymic Relations in French with Knowledge-Pattern-Based Word Sketches.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université du Québec","funders":"","keywords":"Computer science; Sketch; Natural language processing; Artificial intelligence; Word (group theory); Grammar; Domain (mathematical analysis); Thesaurus; Process (computing); Information extraction; Information retrieval; Linguistics","authors":[{"name":"Antonio San Martín","is_ca":true},{"name":"Catherine Trekker","is_ca":false},{"name":"Pilar León-Araúz","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02114258455556268,"gpt":0.2977898512918377,"spread":0.276647266736275,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006947026,0.000888418,0.0005514081,0.005005981,0.0006419041,0.001660481,0.0006493345,0.0007744145,0.008609612],"category_scores_gemma":[0.005581913,0.0002749974,0.0007662004,0.002249846,0.0003843728,0.002743361,0.001027728,0.0006437101,0.003558904],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006369454,"about_ca_system_score_gemma":0.00136964,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01662618,"about_ca_topic_score_gemma":0.01862292,"domain_scores_codex":[0.9991581,0.0002449134,0.0001048009,0.0002825802,0.0001532215,0.00005640501],"domain_scores_gemma":[0.9971107,0.001794297,0.0001823812,0.0002181132,0.0006017523,0.00009283762],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009723075,0.0002692672,0.01639276,0.003049782,0.0002578172,0.001692395,0.003133205,0.005318453,0.0839956,0.01497214,0.02510395,0.8448423],"study_design_scores_gemma":[0.0006292653,0.001211783,0.1021965,0.00105131,0.001196101,0.006604182,0.01250419,0.3168493,0.1583902,0.04515303,0.3538961,0.0003180351],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5018258,0.006021926,0.3819175,0.001083235,0.0003322148,0.001266956,0.05447997,0.02993334,0.02313905],"genre_scores_gemma":[0.6395427,0.00142698,0.29002,0.0001351384,0.00008488107,0.0004024233,0.06036721,0.0009240129,0.007096636],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01662618,"threshold_uncertainty_score":0.03305882,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3029985558","doi":"","title":"The Nunavut Hansard Inuktitut-English Parallel Corpus 3.0 with Preliminary Machine Translation Results.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Computer science; Sentence; Machine translation; Indigenous; Natural language processing; Indigenous language; Linguistics; Artificial intelligence; Speech recognition","authors":[{"name":"Eric Joanis","is_ca":false},{"name":"Rebecca Knowles","is_ca":false},{"name":"Roland Kühn","is_ca":false},{"name":"Samuel Larkin","is_ca":false},{"name":"Patrick Littell","is_ca":false},{"name":"Chi-kiu Lo","is_ca":false},{"name":"Darlene Stewart","is_ca":false},{"name":"Jeffrey Micher","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02149560907502406,"gpt":0.266804596939113,"spread":0.2453089878640889,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003264385,0.001492224,0.001309856,0.004440198,0.003944851,0.002618596,0.001898297,0.001098395,0.04886709],"category_scores_gemma":[0.01071048,0.0007707741,0.0005438368,0.004783424,0.0009296555,0.002248779,0.004059492,0.001328,0.02704414],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001799707,"about_ca_system_score_gemma":0.006332908,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.07243703,"about_ca_topic_score_gemma":0.0986783,"domain_scores_codex":[0.9966558,0.001575168,0.0002603417,0.0006720349,0.0005477293,0.0002888702],"domain_scores_gemma":[0.995404,0.001217031,0.0001329467,0.0007640434,0.002080471,0.0004015454],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00447803,0.001147192,0.006520201,0.005194644,0.000416802,0.002465921,0.005251723,0.004453098,0.03537055,0.01467373,0.6433319,0.2766962],"study_design_scores_gemma":[0.001820175,0.0006039712,0.03017794,0.00100806,0.0005303608,0.002277365,0.004137444,0.01249416,0.04527503,0.006427313,0.8949913,0.0002567512],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.2388603,0.01079709,0.04508836,0.003389696,0.002971737,0.003774299,0.4888813,0.0269908,0.1792465],"genre_scores_gemma":[0.1933154,0.001514824,0.08629465,0.0005749202,0.0002604269,0.002897145,0.6639236,0.006602052,0.04461699],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.07243703,"threshold_uncertainty_score":0.1634766,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4393949820","doi":"10.1007/s10579-024-09720-4","title":"Depression symptoms modelling from social media text: an LLM driven semi-supervised learning approach","year":2024,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Mental Health via Writing","field":"Psychology","cited_by":28,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs; Alberta Machine Intelligence Institute","keywords":"Social media; Depression (economics); Psychology; Artificial intelligence; Computer science; Cognitive psychology; Natural language processing; World Wide Web","authors":[{"name":"Nawshad Farruque","is_ca":true},{"name":"Randy Goebel","is_ca":true},{"name":"Sudhakar Sivapalan","is_ca":true},{"name":"Osmar R. Zai͏̈ane","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05399665033771572,"gpt":0.3638496073975532,"spread":0.3098529570598375,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003354507,0.0009379453,0.001232853,0.00110831,0.0004915761,0.00117265,0.003298184,0.001777692,0.001507229],"category_scores_gemma":[0.008420451,0.0006858988,0.001161354,0.0007575578,0.0007871129,0.001272599,0.001603645,0.00214031,0.001039598],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001032726,"about_ca_system_score_gemma":0.001151212,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005066636,"about_ca_topic_score_gemma":0.007454724,"domain_scores_codex":[0.9982721,0.0007351019,0.0001459113,0.0004739781,0.0002441538,0.0001287473],"domain_scores_gemma":[0.9913669,0.005956281,0.0006312205,0.0006066253,0.001192569,0.0002464268],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000624489,0.00127349,0.01404242,0.0004939333,0.0003547632,0.0005556996,0.0006839805,0.5278651,0.007538118,0.004422009,0.01345886,0.4286871],"study_design_scores_gemma":[0.00000813917,0.0000267113,0.0002600827,0.000005503316,0.000006321032,0.00001505342,0.00001592604,0.99713,0.0006462433,0.001687035,0.0001926424,0.000006336628],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09471881,0.0005388293,0.8971083,0.0009710612,0.00009026416,0.0003164296,0.001020205,0.004127839,0.001108202],"genre_scores_gemma":[0.7661081,0.0001251507,0.2265061,0.0006267092,0.0001734991,0.0004379642,0.002842323,0.0002227957,0.002957289],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005066636,"threshold_uncertainty_score":0.01774055,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4285793742","doi":"10.1007/s10579-022-09602-7","title":"Speech acts in the Dutch COVID-19 Press Conferences","year":2022,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Text Readability and Simplification","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek; Canadian Institute of Steel Construction","keywords":"Speech act; Computer science; Metadata; Linguistics; Reciprocal; Coronavirus disease 2019 (COVID-19); Direct speech; Natural language processing; Mean reciprocal rank; Artificial intelligence; Classifier (UML); Transformer; British National Corpus; World Wide Web; Medicine; Philosophy","authors":[{"name":"Daan Schueler","is_ca":false},{"name":"M. Marx","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07035566898821673,"gpt":0.3433580144822387,"spread":0.273002345494022,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001915436,0.0007202451,0.0004916665,0.002778116,0.0008458489,0.001682085,0.0006636202,0.0006525142,0.009995425],"category_scores_gemma":[0.008203429,0.0003265139,0.0003328216,0.003380551,0.0007331179,0.001162093,0.001463239,0.000691821,0.004089571],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001860043,"about_ca_system_score_gemma":0.001200853,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02423483,"about_ca_topic_score_gemma":0.02685445,"domain_scores_codex":[0.9961696,0.001540491,0.0004845132,0.0005978607,0.00101289,0.0001946486],"domain_scores_gemma":[0.9933873,0.0036737,0.0007491782,0.0003809411,0.001501318,0.0003076277],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.002975809,0.001144042,0.07848129,0.01608176,0.0002850629,0.006679327,0.06905492,0.007103841,0.05590142,0.009845426,0.3526373,0.3998097],"study_design_scores_gemma":[0.0002842096,0.000253843,0.402066,0.001566417,0.0001395315,0.002421668,0.01994064,0.0105602,0.01650351,0.001504969,0.5445036,0.0002555291],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6753801,0.004962876,0.009906097,0.0007445315,0.0004772217,0.001155518,0.2691728,0.0006884756,0.03751247],"genre_scores_gemma":[0.5628696,0.001480236,0.01297132,0.0001915398,0.0001767247,0.002983246,0.4042403,0.0005168712,0.01457024],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02423483,"threshold_uncertainty_score":0.04818755,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3030307641","doi":"","title":"An Analysis of Massively Multilingual Neural Machine Translation for Low-Resource Languages.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Artificial intelligence; Set (abstract data type); Resource (disambiguation); Massively parallel; Programming language","authors":[{"name":"Aaron Mueller","is_ca":false},{"name":"Garrett Nicolai","is_ca":true},{"name":"Arya D. McCarthy","is_ca":false},{"name":"Dylan Lewis","is_ca":false},{"name":"Winston Wu","is_ca":false},{"name":"David Yarowsky","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02852958433192382,"gpt":0.3391111709409689,"spread":0.3105815866090451,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003244531,0.0006848348,0.000626541,0.001489123,0.000847617,0.001275723,0.001027538,0.0007936074,0.003809583],"category_scores_gemma":[0.01465701,0.0002604413,0.0004849064,0.001756445,0.0005013322,0.002046158,0.0008248892,0.0007674697,0.001191543],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001282894,"about_ca_system_score_gemma":0.001214349,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008634785,"about_ca_topic_score_gemma":0.01364109,"domain_scores_codex":[0.9980459,0.0009725863,0.0001147329,0.0002076677,0.0004979878,0.0001610464],"domain_scores_gemma":[0.990813,0.006304179,0.0002697719,0.000719206,0.001759858,0.0001340282],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004753575,0.001309321,0.02244006,0.002167287,0.0009268579,0.002007067,0.0006473149,0.3705645,0.03284,0.03933513,0.05598141,0.4670275],"study_design_scores_gemma":[0.00005262333,0.0002424888,0.006318212,0.00003834103,0.0001063825,0.0002477912,0.0001789737,0.9655091,0.00983749,0.0135077,0.003937971,0.0000229325],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.812707,0.006692375,0.1366827,0.002483083,0.0003927993,0.0003296568,0.006200648,0.008270397,0.02624139],"genre_scores_gemma":[0.9479988,0.0004999817,0.04006866,0.0001668362,0.00009550575,0.0001320127,0.006889428,0.0004221512,0.003726591],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008634785,"threshold_uncertainty_score":0.01716906,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3122399969","doi":"10.1007/s10579-020-09519-z","title":"Exploring the role of lexis and grammar for the stable identification of register in an unrestricted corpus of web documents","year":2021,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"McMaster University; Brockhouse Institute for Materials Research; Brock University","funders":"National Science Foundation of Sri Lanka; Social Sciences and Humanities Research Council of Canada; Turun Yliopisto; Emil Aaltosen Säätiö; AGE-WELL; National Science Foundation","keywords":"Lexis; Register (sociolinguistics); Computer science; Natural language processing; Variation (astronomy); Artificial intelligence; Corpus linguistics; Identification (biology); Linguistics; Text corpus","authors":[{"name":"Veronika Laippala","is_ca":false},{"name":"Jesse Egbert","is_ca":false},{"name":"Douglas Biber","is_ca":false},{"name":"Aki-Juhani Kyröläinen","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05024759076246877,"gpt":0.3165752133910301,"spread":0.2663276226285614,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009382294,0.0008739304,0.0008480606,0.003522418,0.001760971,0.005902129,0.001311682,0.001412116,0.00243498],"category_scores_gemma":[0.04574421,0.0007533643,0.00131153,0.002167539,0.003206459,0.009930177,0.002618184,0.003651771,0.001171437],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001962395,"about_ca_system_score_gemma":0.002071826,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01359849,"about_ca_topic_score_gemma":0.02087571,"domain_scores_codex":[0.9966619,0.002075292,0.0001680064,0.0007769508,0.0002063381,0.0001114537],"domain_scores_gemma":[0.9683222,0.02622248,0.001470128,0.002487725,0.001107539,0.0003898765],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001468755,0.000623674,0.2460164,0.0008695507,0.0009639168,0.001304475,0.01427177,0.2193172,0.01552687,0.1178651,0.01175561,0.3700167],"study_design_scores_gemma":[0.00004172297,0.0000882754,0.01661764,0.0001138218,0.0001204338,0.0002028382,0.001393311,0.8837363,0.002129802,0.09122659,0.004254509,0.00007474152],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7490824,0.002278871,0.2282239,0.005804178,0.0001340149,0.0002114759,0.001898811,0.001328064,0.01103816],"genre_scores_gemma":[0.9544843,0.0004414921,0.04089524,0.000209798,0.00006928164,0.0001268929,0.002146523,0.000320802,0.001305709],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01359849,"threshold_uncertainty_score":0.0496189,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3032216439","doi":"","title":"Fine-grained Morphosyntactic Analysis and Generation Tools for More Than One Thousand Languages.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Natural language processing; Security token; Metric (unit); Artificial intelligence; Machine translation; Parallel corpora; Linguistics; Engineering","authors":[{"name":"Garrett Nicolai","is_ca":true},{"name":"Dylan Lewis","is_ca":false},{"name":"Arya D. McCarthy","is_ca":false},{"name":"Aaron Mueller","is_ca":false},{"name":"Winston Wu","is_ca":false},{"name":"David Yarowsky","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04463166772679519,"gpt":0.3218636676018755,"spread":0.2772319998750803,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00396432,0.001301012,0.000754406,0.002481219,0.0008222492,0.001749112,0.001954132,0.001090645,0.01313722],"category_scores_gemma":[0.01040028,0.0007572334,0.001064951,0.002270422,0.0006920313,0.004430672,0.003098577,0.001390174,0.006868721],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009017299,"about_ca_system_score_gemma":0.001854258,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005240927,"about_ca_topic_score_gemma":0.006514921,"domain_scores_codex":[0.9967332,0.001000874,0.0004056922,0.0006314017,0.001021888,0.0002070902],"domain_scores_gemma":[0.9929865,0.002945157,0.0002443474,0.002053443,0.00144296,0.0003276266],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0015035,0.0007974874,0.007414925,0.001710637,0.0004393243,0.0006712041,0.001655701,0.01178504,0.07274408,0.01887079,0.1155665,0.7668409],"study_design_scores_gemma":[0.001269427,0.001377503,0.0238028,0.0006670393,0.000718726,0.001630258,0.002107222,0.2468246,0.2704061,0.06490397,0.38587,0.0004223395],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1806698,0.00363957,0.5568027,0.001218331,0.0006891248,0.001402364,0.04417009,0.1833752,0.02803276],"genre_scores_gemma":[0.3336333,0.0007437355,0.549821,0.0004078678,0.00009308986,0.001096809,0.09082494,0.009916182,0.01346307],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01313722,"threshold_uncertainty_score":0.04394841,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2896826757","doi":"10.1007/s10579-018-9430-2","title":"VERTa: a linguistic approach to automatic machine translation evaluation","year":2018,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Ministerio de Asuntos Económicos y Transformación Digital, Gobierno de España; Alberta-Pacific Forest Industries","keywords":"Metric (unit); Computer science; Machine translation; Natural language processing; Variety (cybernetics); Artificial intelligence; Linguistics; Engineering","authors":[{"name":"Elisabet Comelles","is_ca":false},{"name":"Jordi Atserias","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03038743013763573,"gpt":0.3296793279805589,"spread":0.2992918978429232,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01327602,0.002232209,0.00212241,0.008677003,0.00236463,0.006756734,0.003678476,0.002317897,0.01080819],"category_scores_gemma":[0.03083114,0.001215867,0.00154607,0.004650963,0.001393766,0.006629603,0.005087656,0.002953477,0.00554728],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001554767,"about_ca_system_score_gemma":0.003558728,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004772209,"about_ca_topic_score_gemma":0.008031053,"domain_scores_codex":[0.968594,0.02035155,0.002028133,0.001913364,0.006417021,0.0006959033],"domain_scores_gemma":[0.9791027,0.009897619,0.0007605152,0.003166916,0.00660485,0.0004674677],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001363081,0.001032446,0.002972124,0.001997556,0.0009867551,0.0004803512,0.001155837,0.0178584,0.04439858,0.04210081,0.1049659,0.7806882],"study_design_scores_gemma":[0.0006484146,0.001183056,0.003998065,0.000476478,0.0007216603,0.001056969,0.001118432,0.7118655,0.08275644,0.07799259,0.1177626,0.0004198474],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0222511,0.001412153,0.8927895,0.0006326675,0.0006652174,0.001506054,0.00542806,0.05830555,0.0170097],"genre_scores_gemma":[0.1394438,0.0006178573,0.8248838,0.0004997357,0.000318033,0.00167253,0.01448422,0.007433807,0.01064623],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01327602,"threshold_uncertainty_score":0.07021117,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1921587136","doi":"10.1007/s10579-015-9318-3","title":"Cross level semantic similarity: an evaluation framework for universal measures of similarity","year":2015,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"European Research Council","keywords":"Computer science; Natural language processing; Similarity (geometry); Semantic similarity; Sentence; Task (project management); Artificial intelligence; Word (group theory); SemEval; Paragraph; Process (computing); Meaning (existential); WordNet; Information retrieval; Linguistics; Psychology; World Wide Web","authors":[{"name":"David Jurgens","is_ca":true},{"name":"Mohammad Taher Pilehvar","is_ca":false},{"name":"Roberto Navigli","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.212328201095616,"gpt":0.3869887300338786,"spread":0.1746605289382627,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02412921,0.001391183,0.00207106,0.01212362,0.001430447,0.005369069,0.002473444,0.002148517,0.002995464],"category_scores_gemma":[0.06528036,0.0005231361,0.001974848,0.007821052,0.001956835,0.01136366,0.005723787,0.002126443,0.0008729197],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001951346,"about_ca_system_score_gemma":0.001957381,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001939076,"about_ca_topic_score_gemma":0.001862738,"domain_scores_codex":[0.9749388,0.01090835,0.002760674,0.002648959,0.008026661,0.0007165587],"domain_scores_gemma":[0.9645081,0.01954816,0.002398135,0.006381042,0.005970902,0.001193555],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001907163,0.0009123082,0.02824858,0.001678404,0.001776341,0.0002376801,0.002643757,0.02845666,0.01970136,0.1872264,0.01001344,0.7171978],"study_design_scores_gemma":[0.0002019675,0.002026813,0.02248336,0.0005611283,0.001246642,0.001179397,0.001584848,0.5581739,0.02920008,0.3625684,0.02039169,0.000381724],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02168853,0.001003108,0.9720469,0.0001430849,0.00008880923,0.0004102549,0.0006332746,0.001345997,0.002640011],"genre_scores_gemma":[0.4087933,0.0005545408,0.5853214,0.0001463601,0.0001934645,0.001135799,0.001917916,0.0005842705,0.001353115],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02412921,"threshold_uncertainty_score":0.127609,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2582792602","doi":"","title":"Detecting semantic changes in Alzheimer’s disease with vector space models","year":2016,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":13,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Disease; Space (punctuation); Medicine; Pathology","authors":[{"name":"Kathleen Fraser","is_ca":true},{"name":"Graeme Hirst","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0302507815983099,"gpt":0.2936973496375037,"spread":0.2634465680391938,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003042013,0.0009370865,0.0008403615,0.004642628,0.0004632439,0.00158138,0.0008128916,0.000952925,0.0009940042],"category_scores_gemma":[0.0072942,0.0001725117,0.001288676,0.002417158,0.0004052613,0.001989576,0.0008062003,0.0007334138,0.0003595836],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001136625,"about_ca_system_score_gemma":0.001154074,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02035109,"about_ca_topic_score_gemma":0.01216657,"domain_scores_codex":[0.9983979,0.0006083999,0.0001886971,0.0003024856,0.0003690199,0.0001334515],"domain_scores_gemma":[0.9959372,0.003069145,0.0002144327,0.0001829804,0.0004995887,0.00009658597],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002804148,0.001222556,0.0711288,0.0005906096,0.001054737,0.0005742395,0.0003539895,0.2114805,0.008699782,0.005788444,0.009770092,0.686532],"study_design_scores_gemma":[0.00004657423,0.0002215058,0.005412978,0.00003114212,0.0001728035,0.0001706221,0.0001742124,0.9824923,0.003475711,0.007036544,0.0007468592,0.00001868418],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7153819,0.003203764,0.2683927,0.001206295,0.0001686967,0.0002947983,0.005483467,0.003967141,0.001901144],"genre_scores_gemma":[0.9340327,0.0005317718,0.05935879,0.0001326485,0.00004448018,0.0001204473,0.004875879,0.00008048608,0.0008228886],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02035109,"threshold_uncertainty_score":0.04046524,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2250536124","doi":"","title":"Discovering frames in specialized domains","year":2014,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal","funders":"","keywords":"FrameNet; Computer science; Complement (music); Field (mathematics); Artificial intelligence; Natural language processing; Information retrieval; Parsing; Mathematics","authors":[{"name":"Marie-Claude L’Homme","is_ca":true},{"name":"Beno it Robichaud","is_ca":true},{"name":"Carlos Subirats Rüggeberg","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01139944083944946,"gpt":0.2946918213716688,"spread":0.2832923805322194,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002623223,0.001334684,0.001575658,0.00669073,0.001688905,0.003823603,0.002301669,0.002031648,0.007951257],"category_scores_gemma":[0.01424785,0.0007010899,0.001619433,0.003815797,0.001148997,0.00955022,0.002934509,0.001833174,0.002217483],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001808833,"about_ca_system_score_gemma":0.002633616,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01332579,"about_ca_topic_score_gemma":0.02022315,"domain_scores_codex":[0.9961202,0.001258194,0.0002852331,0.001035269,0.000810688,0.0004903987],"domain_scores_gemma":[0.9907265,0.005545003,0.0004317174,0.001531533,0.001320764,0.0004445499],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002561462,0.001287682,0.0314226,0.001002252,0.0004590097,0.001848875,0.002120688,0.03211481,0.03084171,0.07739259,0.0236104,0.7953379],"study_design_scores_gemma":[0.0002988354,0.0005231222,0.009576034,0.0002966812,0.0007076116,0.000822683,0.004203866,0.7242944,0.05807185,0.1743076,0.02678604,0.0001112781],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4101197,0.002058741,0.5589414,0.00146152,0.0001550948,0.0006675455,0.005838694,0.006687674,0.01406956],"genre_scores_gemma":[0.7084832,0.0009271608,0.2723902,0.0002406809,0.0001054113,0.0001982729,0.01210873,0.0006971947,0.004849269],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01332579,"threshold_uncertainty_score":0.02659959,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3032025286","doi":"","title":"Cifu: a Frequency Lexicon of Hong Kong Cantonese.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Linguistic Variation and Morphology","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Lexicon; Computer science; Lexical database; Word lists by frequency; Natural language processing; Lexical diversity; Artificial intelligence; Linguistics; Word (group theory); Phonology; Speech recognition; Vocabulary","authors":[{"name":"Regine Lai","is_ca":false},{"name":"Grégoire Winterstein","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0484546324202565,"gpt":0.3496756248661653,"spread":0.3012209924459088,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007203024,0.001029606,0.0004768316,0.005183459,0.001328657,0.001685167,0.0007218004,0.0002851599,0.01379566],"category_scores_gemma":[0.00275724,0.0002790742,0.0002257788,0.005367296,0.000497779,0.001517551,0.0009699404,0.0003844304,0.002156494],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002684825,"about_ca_system_score_gemma":0.004822945,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.2457548,"about_ca_topic_score_gemma":0.2203944,"domain_scores_codex":[0.9996526,0.00007433435,0.00009531876,0.00006451681,0.00007331763,0.00003993783],"domain_scores_gemma":[0.9983336,0.0003899707,0.0001152934,0.000173135,0.0008681649,0.0001197609],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001007124,0.0002076141,0.1042228,0.004230725,0.0003074145,0.002820788,0.03421256,0.003625031,0.03339489,0.03629917,0.2434427,0.5362292],"study_design_scores_gemma":[0.0001178484,0.0001879266,0.3539337,0.0006457585,0.0004067233,0.002419744,0.01381011,0.009488307,0.008754553,0.00346575,0.606543,0.0002265708],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"other","genre_scores_codex":[0.5587788,0.004032266,0.04017689,0.000728969,0.0003660244,0.001185789,0.2705348,0.005719889,0.1184766],"genre_scores_gemma":[0.8344222,0.001026194,0.02854934,0.0001155358,0.00004821115,0.0008829161,0.1152545,0.0006968945,0.01900418],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.2457548,"threshold_uncertainty_score":0.4886487,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3032201847","doi":"","title":"SpiCE: A New Open-Access Corpus of Conversational Bilingual Speech in Cantonese and English.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Transcription (linguistics); Spice; Sentence; Annotation; Natural language processing; Speech corpus; Storyboard; Phonetic transcription; Speech recognition; Linguistics; Artificial intelligence; Speech synthesis; Multimedia; Engineering","authors":[{"name":"Khia A. Johnson","is_ca":true},{"name":"Molly Babel","is_ca":true},{"name":"Ivan Fong","is_ca":false},{"name":"Nancy Yiu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04815606913529093,"gpt":0.3652228133420278,"spread":0.3170667442067369,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.001410597,0.001101486,0.0007409334,0.003325559,0.001792832,0.001402179,0.001320241,0.000876927,0.01745474],"category_scores_gemma":[0.006742451,0.0003459453,0.000297857,0.002779855,0.0008624537,0.001621078,0.002911014,0.001002546,0.004922333],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001167908,"about_ca_system_score_gemma":0.003678437,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.06782387,"about_ca_topic_score_gemma":0.1025796,"domain_scores_codex":[0.9985657,0.0004423434,0.0001823456,0.0003260995,0.0003210031,0.0001624493],"domain_scores_gemma":[0.9952591,0.001540232,0.0002556547,0.0006179137,0.001774386,0.0005526578],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.003967754,0.001401783,0.04658138,0.007127294,0.0004079627,0.00424291,0.01999146,0.00279955,0.1252282,0.008758069,0.4309106,0.3485832],"study_design_scores_gemma":[0.0009013352,0.0005803974,0.3678876,0.001103859,0.0004040513,0.003614065,0.01548167,0.009502767,0.02773397,0.003194288,0.5692571,0.0003388344],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.4031602,0.003070674,0.01520615,0.001127867,0.0005689238,0.001671395,0.5331783,0.005060051,0.03695647],"genre_scores_gemma":[0.3381445,0.0006790278,0.01744675,0.000255834,0.0001597613,0.00265277,0.627647,0.0008236541,0.01219058],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9986798,"threshold_uncertainty_score":0.1348582,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3030151190","doi":"","title":"Evaluating Approaches to Personalizing Language Models","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of New Brunswick","funders":"","keywords":"Perplexity; Computer science; Language model; Personalization; Artificial intelligence; Natural language processing; Adaptation (eye); Word (group theory); World Wide Web","authors":[{"name":"Milton King","is_ca":true},{"name":"Paul Cook","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3079769636386057,"gpt":0.3464064700721486,"spread":0.03842950643354293,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04093151,0.002124774,0.001675756,0.004737976,0.001321046,0.004565788,0.002397554,0.003610964,0.002005492],"category_scores_gemma":[0.127417,0.0009409156,0.00164721,0.0028031,0.001466242,0.008082716,0.003337561,0.003214137,0.0006081397],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003403644,"about_ca_system_score_gemma":0.002988666,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00896824,"about_ca_topic_score_gemma":0.01218979,"domain_scores_codex":[0.9626303,0.02760296,0.001848835,0.00280948,0.004513926,0.0005944825],"domain_scores_gemma":[0.7904698,0.1910407,0.002701092,0.008089342,0.005878296,0.001820724],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006334724,0.003655672,0.03196119,0.001922498,0.002634977,0.0001614866,0.002927196,0.2321475,0.005953657,0.01322066,0.007822404,0.691258],"study_design_scores_gemma":[0.0006525703,0.001410466,0.004714962,0.0001928941,0.001468409,0.0001196378,0.0008012542,0.9546984,0.007566678,0.02566815,0.002595341,0.0001110755],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4724972,0.01110287,0.4950416,0.002418332,0.0003952516,0.001810054,0.001202132,0.005039508,0.01049305],"genre_scores_gemma":[0.8015619,0.001617974,0.1918664,0.0003511941,0.000171298,0.0006186442,0.001671994,0.0003203298,0.001820252],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04093151,"threshold_uncertainty_score":0.2164691,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3029096252","doi":"","title":"A Lexicon-Based Approach for Detecting Hedges in Informal Text","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Western University","funders":"","keywords":"Computer science; Hedge; Lexicon; Natural language processing; Sentence; Interview; Artificial intelligence; Part-of-speech tagging; Linguistics; Part of speech; Sociology","authors":[{"name":"Jumayel Islam","is_ca":true},{"name":"Lu Xiao","is_ca":false},{"name":"Robert E. Mercer","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03137950489801756,"gpt":0.300703666162043,"spread":0.2693241612640255,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002935792,0.001012575,0.001150095,0.01020302,0.001259121,0.004010528,0.001445174,0.001627269,0.004484517],"category_scores_gemma":[0.0138616,0.0005037769,0.0008137898,0.003971232,0.0008862177,0.004704064,0.002353044,0.001103306,0.002264699],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001109881,"about_ca_system_score_gemma":0.002365815,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006771499,"about_ca_topic_score_gemma":0.01154917,"domain_scores_codex":[0.9951448,0.001625462,0.0006684294,0.0006234715,0.001699782,0.0002380165],"domain_scores_gemma":[0.9887449,0.005880194,0.0007561448,0.001021963,0.003175638,0.0004213011],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008186192,0.0008050266,0.01893375,0.001071718,0.0003042013,0.001050859,0.001876044,0.006085202,0.09664157,0.02304707,0.02209923,0.8272668],"study_design_scores_gemma":[0.0003625563,0.001032251,0.02632898,0.00042108,0.0007855103,0.002580915,0.003098581,0.7522784,0.1098275,0.05682352,0.04605179,0.0004088343],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1103757,0.001223971,0.8496416,0.0006671164,0.000166984,0.00177843,0.0057747,0.0191111,0.01126026],"genre_scores_gemma":[0.4150704,0.0003425179,0.5716984,0.00022316,0.00008531076,0.000606762,0.007390611,0.0006437546,0.003939128],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01020302,"threshold_uncertainty_score":0.01552618,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3031229455","doi":"","title":"SEDAR: a Large Scale French-English Financial Domain Parallel Corpus","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal","funders":"","keywords":"Machine translation; Computer science; Domain (mathematical analysis); Preprocessor; Natural language processing; Sentence; Translation (biology); Artificial intelligence; Scale (ratio); Speech recognition; Chemistry","authors":[{"name":"Abbas Ghaddar","is_ca":false},{"name":"Phillippe Langlais","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01198234994470673,"gpt":0.2627608228998279,"spread":0.2507784729551212,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00177393,0.001241071,0.0007534519,0.00438736,0.00212779,0.001677125,0.0014399,0.001433059,0.0222195],"category_scores_gemma":[0.007047587,0.0004436807,0.0006326935,0.003111232,0.001014873,0.001985593,0.001856933,0.001323009,0.009744931],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001455152,"about_ca_system_score_gemma":0.003420402,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04281013,"about_ca_topic_score_gemma":0.04490782,"domain_scores_codex":[0.9985623,0.000552464,0.0001315275,0.0003325605,0.0002925288,0.0001287136],"domain_scores_gemma":[0.9950516,0.002121869,0.0001736339,0.0005859228,0.001739104,0.000327863],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002438765,0.001536986,0.01375937,0.004503166,0.0005985738,0.00497691,0.003115494,0.008787898,0.05996235,0.02116169,0.6064276,0.2727313],"study_design_scores_gemma":[0.001291356,0.0005878602,0.07005242,0.0004518992,0.0005377599,0.004406522,0.00410619,0.03391825,0.03335688,0.008594587,0.8423951,0.0003011929],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.3599276,0.004966531,0.05351289,0.004001608,0.001308697,0.001363764,0.5016189,0.01960641,0.05369358],"genre_scores_gemma":[0.2726698,0.001143492,0.06100544,0.0008267665,0.0003155982,0.001221858,0.6451708,0.002067097,0.01557907],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.04281013,"threshold_uncertainty_score":0.08512193,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2577349305","doi":"","title":"Training & Quality Assessment of an Optical Character Recognition Model for Northern Haida.","year":2016,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Handwritten Text Recognition Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Optical character recognition; Computer science; Character (mathematics); Language model; Unicode; Porting; Artificial intelligence; Natural language processing; Hidden Markov model; Speech recognition; Set (abstract data type); Image (mathematics)","authors":[{"name":"Isabell Hubert Lyall","is_ca":false},{"name":"Antti Arppe","is_ca":true},{"name":"Jordan Lachler","is_ca":true},{"name":"Eddie Antonio Santos","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1226647332271734,"gpt":0.3879749591543143,"spread":0.265310225927141,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00138039,0.000733847,0.0004636255,0.0004987013,0.0006773091,0.0008175591,0.0009798482,0.0006614859,0.002599166],"category_scores_gemma":[0.003678292,0.0003221594,0.0004079424,0.0004174708,0.0003164695,0.0008630855,0.0005934858,0.0005287091,0.001437094],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001306263,"about_ca_system_score_gemma":0.001948607,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.1402283,"about_ca_topic_score_gemma":0.1447897,"domain_scores_codex":[0.9994224,0.0001029505,0.00004370331,0.0001958385,0.0001718022,0.00006333779],"domain_scores_gemma":[0.9979572,0.0004228585,0.00008995333,0.0002519728,0.001179299,0.00009867576],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001737791,0.0006942674,0.04380491,0.0002916771,0.0002456561,0.0003560559,0.0004470484,0.1879176,0.07649422,0.0004741852,0.01272178,0.6748149],"study_design_scores_gemma":[0.0000626551,0.0003828449,0.03711302,0.00002407442,0.0001001638,0.00008860001,0.0003164276,0.9130394,0.04581007,0.0001794591,0.002852395,0.00003086968],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.9339243,0.0005377874,0.05351438,0.0002412699,0.0001824486,0.0002800371,0.00129118,0.004764598,0.00526398],"genre_scores_gemma":[0.961786,0.0001196065,0.02717968,0.00005817198,0.00001417364,0.00008391437,0.003011402,0.0002189084,0.007528018],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.1402283,"threshold_uncertainty_score":0.2788241,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3029265130","doi":"","title":"Multilingual Dictionary Based Construction of Core Vocabulary.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Vocabulary; Natural language processing; Core (optical fiber); Artificial intelligence; Set (abstract data type); Bilingual dictionary; Field (mathematics); Resource (disambiguation); Machine translation; Linguistics; Programming language","authors":[{"name":"Winston Wu","is_ca":false},{"name":"Garrett Nicolai","is_ca":true},{"name":"David Yarowsky","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02869467135172208,"gpt":0.3027328626012327,"spread":0.2740381912495106,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002659374,0.0006668834,0.0008988634,0.004443281,0.001119778,0.002789414,0.001350765,0.0006210512,0.01448565],"category_scores_gemma":[0.01356747,0.0004223374,0.0006385333,0.002629001,0.000666392,0.007674701,0.004650673,0.001208359,0.006870389],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00117116,"about_ca_system_score_gemma":0.003453089,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007330266,"about_ca_topic_score_gemma":0.01202421,"domain_scores_codex":[0.995415,0.001957605,0.0006137853,0.0008134937,0.0009146866,0.0002854656],"domain_scores_gemma":[0.991949,0.002478808,0.0002621952,0.001129746,0.003834054,0.0003462859],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001080108,0.0006002125,0.00880594,0.002004129,0.0002216377,0.0003888469,0.002459609,0.004986783,0.0610783,0.05857586,0.0412084,0.8185901],"study_design_scores_gemma":[0.0006009321,0.001316769,0.01681088,0.001060948,0.0008471427,0.002781912,0.01071279,0.3054191,0.2696757,0.1078484,0.2825933,0.00033204],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1510089,0.001600201,0.7510712,0.0005559606,0.0004962291,0.002355284,0.02160005,0.01650844,0.05480361],"genre_scores_gemma":[0.4719099,0.0006737193,0.4617526,0.0002099203,0.00007120408,0.001273349,0.04732143,0.002317864,0.01447015],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01448565,"threshold_uncertainty_score":0.04845935,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2251445713","doi":"","title":"Capturing syntactico-semantic regularities among terms: An application of the FrameNet methodology to terminology","year":2012,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Montréal","funders":"","keywords":"FrameNet; Computer science; Terminology; Annotation; Focus (optics); Natural language processing; Artificial intelligence; Point (geometry); Resource (disambiguation); Lexicon; Computational linguistics; Semantics (computer science); Linguistics; Parsing; Programming language","authors":[{"name":"Marie-Claude L’Homme","is_ca":true},{"name":"Janine Pimentel","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03784322549439183,"gpt":0.3385109969376508,"spread":0.300667771443259,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008608997,0.0009333952,0.001612914,0.009330446,0.001935428,0.004716146,0.002197416,0.001293532,0.003047703],"category_scores_gemma":[0.02592189,0.0006896139,0.001542614,0.007895039,0.002554159,0.0137222,0.003695484,0.001669506,0.0005761269],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002759759,"about_ca_system_score_gemma":0.003306474,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01384536,"about_ca_topic_score_gemma":0.01641856,"domain_scores_codex":[0.9930485,0.003087429,0.0007766063,0.001015019,0.001736075,0.0003363722],"domain_scores_gemma":[0.9868073,0.007427548,0.001155709,0.001912595,0.002385685,0.0003112676],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003780581,0.0002127614,0.009640882,0.0007437462,0.0002057478,0.0004488218,0.003538319,0.02434735,0.01104869,0.6117911,0.004861158,0.3327834],"study_design_scores_gemma":[0.00003917238,0.0001062868,0.003533663,0.0002734674,0.0002878773,0.0004040241,0.00167702,0.2925686,0.01103336,0.6658524,0.02411856,0.0001056043],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03556982,0.0005616429,0.9552153,0.0006376135,0.00006405993,0.0003107327,0.001379534,0.0008436173,0.005417747],"genre_scores_gemma":[0.3429803,0.0006445834,0.6509994,0.0001226642,0.00009275464,0.0003823841,0.002801635,0.0004037556,0.001572488],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01384536,"threshold_uncertainty_score":0.04552925,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1990825561","doi":"10.1007/s10579-008-9072-x","title":"Disambiguation of partial cognates","year":2008,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Ottawa","funders":"","keywords":"Cognate; Computer science; Natural language processing; Meaning (existential); Context (archaeology); Artificial intelligence; Word (group theory); Word-sense disambiguation; Machine translation; Linguistics; Psychology; Biology","authors":[{"name":"Oana Frunza","is_ca":true},{"name":"Diana Inkpen","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0240283775658798,"gpt":0.3017247175024333,"spread":0.2776963399365535,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006116039,0.001459735,0.002303403,0.005921434,0.003163668,0.005768831,0.002042032,0.001811421,0.008848233],"category_scores_gemma":[0.02146505,0.0006951427,0.001535985,0.002429568,0.001787734,0.01053067,0.005356506,0.001713185,0.00309744],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001017604,"about_ca_system_score_gemma":0.002856692,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004551423,"about_ca_topic_score_gemma":0.005629841,"domain_scores_codex":[0.9912361,0.003672736,0.00103569,0.001741673,0.001669076,0.0006447629],"domain_scores_gemma":[0.9856762,0.007221584,0.0004057398,0.002683434,0.003360565,0.0006524947],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.004896315,0.0006061334,0.01905972,0.001615557,0.0006600255,0.00166222,0.002497594,0.01793492,0.0508,0.06224063,0.0210597,0.8169671],"study_design_scores_gemma":[0.0005376345,0.0009701094,0.01528149,0.0005511554,0.001881893,0.005065022,0.005389212,0.4316488,0.2522224,0.2108976,0.07501578,0.0005388839],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.6149464,0.00576524,0.3177124,0.001455067,0.0009577194,0.0006895741,0.004876391,0.01208646,0.04151073],"genre_scores_gemma":[0.8598629,0.0007685002,0.125527,0.0002803565,0.0001895238,0.0001334095,0.005487444,0.001206687,0.006544],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.008848233,"threshold_uncertainty_score":0.03234506,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3030267806","doi":"","title":"On The Performance of Time-Pooling Strategies for End-to-End Spoken Language Identification.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Institut National de la Recherche Scientifique","funders":"","keywords":"Pooling; Computer science; Benchmark (surveying); Artificial intelligence; Dimension (graph theory); Representation (politics); Set (abstract data type); Spoken language; Identification (biology); Machine learning; Selection (genetic algorithm); Language model; Natural language processing; Test set; Mathematics","authors":[{"name":"João Monteiro","is_ca":false},{"name":"Jahangir Alam","is_ca":true},{"name":"Tiago H. Falk","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0337936629644121,"gpt":0.2838120272324091,"spread":0.250018364267997,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004529137,0.001791282,0.001248512,0.0009858542,0.0007086286,0.00157012,0.001630695,0.001796612,0.005151122],"category_scores_gemma":[0.01019369,0.0004217785,0.0005805491,0.0005335441,0.0005035213,0.002906414,0.001696088,0.001159256,0.002402563],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006816384,"about_ca_system_score_gemma":0.001427782,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01341703,"about_ca_topic_score_gemma":0.01699069,"domain_scores_codex":[0.9977842,0.0008316992,0.0001840065,0.0003836875,0.0005280907,0.0002883156],"domain_scores_gemma":[0.9956117,0.003131675,0.0001085567,0.0002891232,0.0006953469,0.0001635623],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.008110458,0.0009206165,0.003624865,0.0006567977,0.0008312168,0.0004386011,0.0002534607,0.09053636,0.07906456,0.003246556,0.01219516,0.8001214],"study_design_scores_gemma":[0.0001493183,0.001149657,0.004423629,0.00004311497,0.0002502191,0.0003913434,0.0002690087,0.9104629,0.07875666,0.002014378,0.00202384,0.00006593536],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4480976,0.009066168,0.5152743,0.0007253456,0.0007121211,0.0003708071,0.001971234,0.01085943,0.01292293],"genre_scores_gemma":[0.812991,0.000981932,0.1695939,0.0004383626,0.0001489033,0.0002018616,0.004178287,0.0004376133,0.0110283],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01341703,"threshold_uncertainty_score":0.02667791,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3032097118","doi":"","title":"Multilingual Corpus Creation for Multilingual Semantic Similarity Task","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo; Western University","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Task (project management); Semantic similarity; Similarity (geometry); Sentence; Focus (optics); Information retrieval","authors":[{"name":"Mahtab Ahmed","is_ca":true},{"name":"Chahna Dixit","is_ca":false},{"name":"Robert E. Mercer","is_ca":true},{"name":"Atif Khan","is_ca":true},{"name":"Muhammad Rifayat Samee","is_ca":true},{"name":"Felipe Urra","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04671167661381604,"gpt":0.3280433324245093,"spread":0.2813316558106933,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003499199,0.001863358,0.001428389,0.00639136,0.003177388,0.003068524,0.001677516,0.001565221,0.02194257],"category_scores_gemma":[0.01147154,0.0006827826,0.001415061,0.003813209,0.000722763,0.005681304,0.005613738,0.002350845,0.01044275],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001120555,"about_ca_system_score_gemma":0.003872688,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008061717,"about_ca_topic_score_gemma":0.008850508,"domain_scores_codex":[0.995864,0.001499945,0.0005469307,0.001022632,0.0007505556,0.0003159441],"domain_scores_gemma":[0.9937643,0.002313206,0.0001618377,0.001024896,0.002258626,0.0004771223],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00235374,0.001643274,0.009811264,0.003140762,0.000514995,0.001891071,0.004350916,0.007965107,0.07819764,0.02039156,0.183147,0.6865927],"study_design_scores_gemma":[0.001172926,0.001591998,0.02421838,0.000684777,0.001488182,0.005081438,0.01224594,0.3211801,0.2338276,0.0311179,0.3667325,0.0006581331],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3598339,0.003532889,0.4620368,0.001662439,0.002851008,0.00428038,0.06052716,0.04745438,0.05782105],"genre_scores_gemma":[0.4802123,0.0008964211,0.3586167,0.0004036339,0.0004634451,0.004078203,0.1349199,0.004525758,0.0158835],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02194257,"threshold_uncertainty_score":0.07340527,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3030379882","doi":"","title":"On the Creation of a Corpus for Coherence Evaluation of Discursive Units","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Concordia University; Bank of Canada; National Bank of Canada","funders":"","keywords":"Computer science; Coherence (philosophical gambling strategy); Natural language processing; Artificial intelligence; Sentence; Argument (complex analysis); Focus (optics); Classifier (UML); Linguistics; Textual entailment; Logical consequence; Mathematics","authors":[{"name":"Elham Mohammadi","is_ca":true},{"name":"Timothe Beiko","is_ca":false},{"name":"Leila Kosseim","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05399198577407965,"gpt":0.3305914666254582,"spread":0.2765994808513786,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01233511,0.0007641944,0.000889957,0.006426833,0.003077675,0.003905706,0.002206867,0.001634323,0.009484611],"category_scores_gemma":[0.04772181,0.0007733097,0.0004698477,0.003623722,0.002395862,0.006988534,0.005346199,0.001993589,0.003639278],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001597498,"about_ca_system_score_gemma":0.003278793,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008010175,"about_ca_topic_score_gemma":0.01279062,"domain_scores_codex":[0.9863339,0.008650459,0.00113074,0.00111341,0.002434774,0.00033674],"domain_scores_gemma":[0.9444702,0.03375157,0.001492028,0.005883655,0.01306866,0.001333838],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008691405,0.001313893,0.00950704,0.001427467,0.0001200159,0.000778537,0.011883,0.01145424,0.0596853,0.07109675,0.05890156,0.772963],"study_design_scores_gemma":[0.0007250828,0.001387425,0.03147209,0.001028395,0.0003375647,0.002164219,0.01451746,0.4315942,0.21971,0.07159249,0.2249045,0.0005665644],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2009218,0.0009618101,0.7403103,0.002096623,0.0004093743,0.003829399,0.00969297,0.007953777,0.03382402],"genre_scores_gemma":[0.2721308,0.0003204328,0.7033212,0.0002181776,0.0001199837,0.002702215,0.01305335,0.001549172,0.006584631],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01233511,"threshold_uncertainty_score":0.06523508,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3031309641","doi":"","title":"Temporal Histories of Epidemic Events (THEE): A Case Study in Temporal Annotation for Public Health","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Public Health Agency of Canada; University of Toronto","funders":"","keywords":"Annotation; Computer science; Metadata; Domain (mathematical analysis); Event (particle physics); Public domain; Information retrieval; Process (computing); Temporal annotation; Style (visual arts); Natural language processing; Artificial intelligence; World Wide Web; History; Natural language","authors":[{"name":"Jingcheng Niu","is_ca":true},{"name":"Victoria Ng","is_ca":true},{"name":"Gerald Penn","is_ca":true},{"name":"Erin E. Rees","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1061119531422761,"gpt":0.3841453758342169,"spread":0.2780334226919409,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01120873,0.0004100404,0.0003133568,0.00409508,0.001968867,0.002260977,0.001184744,0.001282897,0.003751104],"category_scores_gemma":[0.03558925,0.0002581252,0.0005320865,0.006504504,0.001316397,0.005319603,0.002411599,0.001296,0.0006549497],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002408865,"about_ca_system_score_gemma":0.004569238,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03867708,"about_ca_topic_score_gemma":0.04608632,"domain_scores_codex":[0.9950819,0.00239803,0.0006978646,0.0005816004,0.001072359,0.0001682381],"domain_scores_gemma":[0.9345662,0.05384064,0.003462475,0.002972795,0.004094421,0.001063554],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001925058,0.0007954218,0.1561126,0.006443338,0.0002696131,0.01413655,0.05409178,0.02357757,0.01513607,0.21557,0.08882712,0.4231149],"study_design_scores_gemma":[0.0001421909,0.0002914542,0.07159671,0.002051718,0.0003682546,0.007154112,0.03912747,0.1253152,0.02255877,0.09470068,0.6364335,0.000259899],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.458118,0.004313724,0.3590918,0.02339149,0.0005723654,0.001803517,0.08893823,0.005998227,0.05777269],"genre_scores_gemma":[0.6997578,0.001776984,0.2527993,0.001084466,0.0001304145,0.0006450691,0.0344471,0.0008909113,0.008468001],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03867708,"threshold_uncertainty_score":0.07690388,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3030811146","doi":"","title":"Language Modeling with a General Second-Order RNN.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Recurrent neural network; Computer science; Multiplicative function; Treebank; Language model; State space; State (computer science); Artificial intelligence; Space (punctuation); Sequence (biology); Artificial neural network; Algorithm; Mathematics; Statistics; Parsing","authors":[{"name":"Diego Maupomé","is_ca":true},{"name":"Marie‐Jean Meurs","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03008292426685108,"gpt":0.2736927433707653,"spread":0.2436098191039142,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003747853,0.001353743,0.0008810738,0.001150607,0.0005457316,0.002317582,0.001990912,0.001817136,0.007478758],"category_scores_gemma":[0.01203705,0.00059592,0.001220449,0.001052406,0.0004007799,0.003572829,0.001491677,0.002819265,0.005345342],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001122696,"about_ca_system_score_gemma":0.001774111,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01335698,"about_ca_topic_score_gemma":0.01795946,"domain_scores_codex":[0.9975619,0.001358964,0.0001136835,0.0005549208,0.0002461541,0.0001643884],"domain_scores_gemma":[0.9955912,0.003017338,0.0001236113,0.000411178,0.0006947764,0.0001618902],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001030604,0.0003189532,0.002593579,0.00056602,0.000430009,0.0002837379,0.0004130266,0.3607503,0.01318882,0.02178155,0.01718134,0.581462],"study_design_scores_gemma":[0.00001290806,0.0000391416,0.0002253941,0.00001998816,0.00003060393,0.00004272824,0.00002939569,0.9884133,0.002352876,0.007179094,0.00164242,0.00001222156],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01729576,0.0008975226,0.9679602,0.0005039775,0.0002601156,0.0002161963,0.001547145,0.005604928,0.005714217],"genre_scores_gemma":[0.5485517,0.0007129519,0.4244514,0.0005517228,0.0002582036,0.0007269519,0.006289679,0.00122986,0.01722753],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01335698,"threshold_uncertainty_score":0.02655846,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3030512995","doi":"","title":"Evaluating Sub-word Embeddings in Cross-lingual Models.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Word (group theory); Natural language processing; Artificial intelligence; Lexicon; Task (project management); Space (punctuation); Vocabulary; Resource (disambiguation); Linguistics","authors":[{"name":"Ali Hakimi Parizi","is_ca":false},{"name":"Paul Cook","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08551581455221315,"gpt":0.3744345054740225,"spread":0.2889186909218093,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0122037,0.002140032,0.001487624,0.003390763,0.00099368,0.003398265,0.001987271,0.002124491,0.003604553],"category_scores_gemma":[0.03435148,0.0005599243,0.00149783,0.00281437,0.0005946276,0.007443707,0.004014874,0.002938618,0.003304408],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001016868,"about_ca_system_score_gemma":0.001802912,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009764874,"about_ca_topic_score_gemma":0.0131167,"domain_scores_codex":[0.9900303,0.006751821,0.0006454022,0.001359582,0.0008760985,0.0003367616],"domain_scores_gemma":[0.9766827,0.01764223,0.0004062811,0.002190877,0.002520456,0.0005574576],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003117843,0.002157377,0.03240322,0.001380242,0.0039509,0.0004019745,0.001124739,0.1246419,0.007364401,0.006609255,0.0516713,0.7651769],"study_design_scores_gemma":[0.0002509017,0.0007965275,0.005848217,0.0001907071,0.0009837814,0.0002710027,0.001118214,0.9616428,0.007858173,0.01339205,0.007548671,0.0000989734],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5994585,0.01572407,0.3341468,0.002659475,0.002871639,0.0007382498,0.01256885,0.01379878,0.01803357],"genre_scores_gemma":[0.8777215,0.001784895,0.08768651,0.0004930767,0.0004199732,0.0004094363,0.02559002,0.0009744075,0.004920041],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0122037,"threshold_uncertainty_score":0.06454009,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3032420555","doi":"","title":"Cooking Up a Neural-based Model for Recipe Classification","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Concordia University; Bank of Canada; Université du Québec à Montréal; Polytechnique Montréal; National Bank of Canada","funders":"","keywords":"Embedding; Computer science; Task (project management); Artificial intelligence; Layer (electronics); Macro; Artificial neural network; Recipe; Natural language processing; Language model; Deep learning; State (computer science); Machine learning; Pattern recognition (psychology); Algorithm; Engineering","authors":[{"name":"Elham Mohammadi","is_ca":true},{"name":"Nada Naji","is_ca":true},{"name":"Louis Marceau","is_ca":true},{"name":"Marc Queudot","is_ca":true},{"name":"Éric Charton","is_ca":true},{"name":"Leila Kosseim","is_ca":true},{"name":"Marie‐Jean Meurs","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1262190574827426,"gpt":0.3325542575692801,"spread":0.2063352000865376,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001903329,0.0009264647,0.0008371901,0.00105586,0.0006376417,0.001745479,0.001489153,0.001941561,0.005485358],"category_scores_gemma":[0.004249679,0.0005598005,0.001100879,0.0008841479,0.000401279,0.002893666,0.0009404281,0.002841444,0.002618741],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00146388,"about_ca_system_score_gemma":0.001144451,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02049641,"about_ca_topic_score_gemma":0.02676594,"domain_scores_codex":[0.9993526,0.0002077709,0.00004341989,0.0002123417,0.0001048358,0.00007901865],"domain_scores_gemma":[0.9986155,0.0006922549,0.00004197242,0.0001517208,0.0004315446,0.00006698327],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009273139,0.0007059289,0.005541811,0.0001926719,0.0004404125,0.0001278318,0.0001687463,0.3341322,0.01472054,0.009172742,0.01608564,0.6177841],"study_design_scores_gemma":[0.00001109235,0.00003962978,0.0003450548,0.000009184121,0.00003728622,0.0000164592,0.00001345786,0.9937894,0.002321947,0.00287026,0.0005350551,0.00001115028],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1260491,0.001422477,0.8528692,0.002731309,0.0005428247,0.0002931613,0.001439812,0.00626941,0.0083828],"genre_scores_gemma":[0.8122604,0.000601353,0.1665064,0.0006877583,0.0002814848,0.000284517,0.002469722,0.0003468558,0.01656157],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02049641,"threshold_uncertainty_score":0.0407542,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3029996834","doi":"","title":"Corpus of Chinese Dynastic Histories: Gender Analysis over Two Millennia.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Chinese history and philosophy","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Lexicon; Computer science; Dominance (genetics); License; Focus (optics); Semantics (computer science); Linguistics; Natural language processing; History; Classical Chinese; Space (punctuation); Corpus linguistics; Artificial intelligence","authors":[{"name":"Sergey A. Zinin","is_ca":false},{"name":"Yang Xu","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03543237694785305,"gpt":0.3434333187503363,"spread":0.3080009418024832,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001047969,0.0004037064,0.0003172462,0.007342028,0.001881488,0.0008396843,0.0006932876,0.0003010491,0.01098729],"category_scores_gemma":[0.005107099,0.0001640052,0.0001551563,0.01444851,0.0007114673,0.001120226,0.001684008,0.0004161498,0.001835542],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002408003,"about_ca_system_score_gemma":0.006122392,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.1148642,"about_ca_topic_score_gemma":0.1620353,"domain_scores_codex":[0.9995894,0.0000840613,0.0000580987,0.00008718431,0.0001201457,0.00006108245],"domain_scores_gemma":[0.9971975,0.001018547,0.0002970941,0.0003099107,0.0008871655,0.0002897708],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0007947475,0.0001477549,0.1401307,0.004805839,0.0001476953,0.003021221,0.08201265,0.0006237265,0.00935282,0.0319869,0.3378414,0.3891346],"study_design_scores_gemma":[0.00002813734,0.000034289,0.4969927,0.0003981816,0.0001021866,0.0005510287,0.01212575,0.0006628698,0.002490754,0.001465825,0.4851079,0.00004044893],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.478618,0.007927224,0.003962149,0.002237449,0.0004688404,0.0003767746,0.4175583,0.0003756185,0.08847569],"genre_scores_gemma":[0.7250482,0.003032029,0.003477114,0.0002139477,0.0002002737,0.000973646,0.2365594,0.0002380946,0.03025736],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1148642,"threshold_uncertainty_score":0.2283912,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3031043157","doi":"","title":"WEXEA: Wikipedia EXhaustive Entity Annotation","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Annotation; Hyperlink; Information retrieval; Entity linking; Named-entity recognition; Relationship extraction; Natural language processing; Task (project management); Publication; Information extraction; Named entity; Relation (database); Artificial intelligence; World Wide Web; Web page; Knowledge base; Database","authors":[{"name":"Michael Strobl","is_ca":true},{"name":"Amine Trabelsi","is_ca":true},{"name":"Osmar R. Zai͏̈ane","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0339419726941602,"gpt":0.2878343625034358,"spread":0.2538923898092756,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009988223,0.003635436,0.002159588,0.01348401,0.003863646,0.004511585,0.003977229,0.002954405,0.02507942],"category_scores_gemma":[0.03560342,0.001270972,0.001926575,0.006931653,0.001198937,0.00945017,0.008674054,0.002844961,0.01761478],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001459771,"about_ca_system_score_gemma":0.005224796,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03309637,"about_ca_topic_score_gemma":0.04394478,"domain_scores_codex":[0.9860215,0.005578929,0.001630083,0.002650826,0.003257885,0.0008607636],"domain_scores_gemma":[0.9679819,0.01457781,0.000956546,0.007211316,0.00730128,0.001971079],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002646182,0.001844655,0.01007355,0.005221295,0.001330604,0.0008505665,0.001307947,0.006505875,0.01190307,0.005157434,0.6825647,0.2705941],"study_design_scores_gemma":[0.002505266,0.002188492,0.03467909,0.002272159,0.002242471,0.002854786,0.004811017,0.1830381,0.07998212,0.02203852,0.6624891,0.0008989591],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"methods","genre_scores_codex":[0.1237874,0.006735745,0.1369563,0.002288386,0.003138864,0.005723819,0.4786192,0.198221,0.04452935],"genre_scores_gemma":[0.1139518,0.001046636,0.1327395,0.0007503279,0.0002369355,0.003116468,0.7258298,0.008416058,0.01391254],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.03309637,"threshold_uncertainty_score":0.08389902,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4407752103","doi":"10.1007/s10579-025-09813-8","title":"The narratives of war (NoW) corpus of written testimonies of the Russia-Ukraine war","year":2025,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Mental Health via Writing","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"McMaster University","funders":"Social Sciences and Humanities Research Council of Canada; Mitacs; Canada Research Chairs","keywords":"Ukrainian; Documentation; Narrative; Spanish Civil War; History; Psychology; Political science; Linguistics; Law; Computer science","authors":[{"name":"Serhii Zasiekin","is_ca":false},{"name":"Larysa Zasiekina","is_ca":false},{"name":"Emilie Altman","is_ca":true},{"name":"Mariia Hryntus","is_ca":true},{"name":"Victor Kuperman","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02440114564653542,"gpt":0.369499880025237,"spread":0.3450987343787016,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001410593,0.000213381,0.0002361955,0.003244933,0.001668181,0.001229169,0.0004074077,0.0004618004,0.004372532],"category_scores_gemma":[0.008827239,0.0001741884,0.0001131183,0.004236461,0.001240618,0.001122674,0.002136319,0.0005371069,0.0006841075],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001270841,"about_ca_system_score_gemma":0.00184678,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005850304,"about_ca_topic_score_gemma":0.01255442,"domain_scores_codex":[0.998616,0.0006049502,0.0002863169,0.0001670111,0.0002536983,0.00007200735],"domain_scores_gemma":[0.9943978,0.003087032,0.0007440647,0.0007308478,0.0008798006,0.0001605347],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0006380792,0.0002149308,0.03492993,0.007428119,0.00009362151,0.006516591,0.471971,0.001424388,0.02359317,0.03056093,0.09912701,0.3235022],"study_design_scores_gemma":[0.00001646371,0.00005481997,0.09085731,0.001305294,0.00004314946,0.002151061,0.08240339,0.0003121725,0.006199418,0.001489096,0.8151228,0.00004498943],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.8463765,0.01137871,0.007819727,0.002145806,0.0008431053,0.0009887379,0.07718787,0.0001644671,0.05309511],"genre_scores_gemma":[0.9079545,0.006703288,0.01897671,0.0004522323,0.0001989908,0.001413787,0.04863334,0.0002084645,0.01545876],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.005850304,"threshold_uncertainty_score":0.01462758,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3032491810","doi":"","title":"GM-RKB WikiText Error Correction Task and Baselines.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Error detection and correction; Artificial intelligence; Natural language processing; Task (project management); Language model; Ground truth; Speech recognition; Information retrieval; Machine learning; Algorithm","authors":[{"name":"Gabor Melli","is_ca":true},{"name":"Abdelrhman Eldallal","is_ca":false},{"name":"Bassim Lazem","is_ca":false},{"name":"O. Moreira","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02172302327528814,"gpt":0.2949398695453348,"spread":0.2732168462700467,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01059755,0.003535073,0.001942068,0.003318715,0.002181076,0.00368667,0.004606127,0.004208244,0.02160893],"category_scores_gemma":[0.05372028,0.0009634225,0.001313136,0.00275908,0.001208548,0.005665122,0.00727488,0.00477811,0.03326852],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001111886,"about_ca_system_score_gemma":0.003411531,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01392336,"about_ca_topic_score_gemma":0.01772046,"domain_scores_codex":[0.9870745,0.004811199,0.001703967,0.002943043,0.002779353,0.000688035],"domain_scores_gemma":[0.965261,0.01306785,0.001404442,0.01021427,0.008017174,0.002035249],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.005223535,0.003812697,0.007675368,0.004758002,0.0007832855,0.000609575,0.0009531707,0.005744122,0.02525153,0.002009973,0.6445594,0.2986194],"study_design_scores_gemma":[0.007347206,0.004299547,0.09471348,0.001924592,0.001967526,0.004745079,0.003159251,0.1546687,0.1674633,0.01371451,0.5449161,0.001080669],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.2874857,0.007231971,0.08066499,0.003746382,0.007037636,0.006965436,0.2961319,0.212249,0.09848691],"genre_scores_gemma":[0.254823,0.0008243613,0.1135713,0.002143651,0.0005430844,0.006159883,0.5628424,0.01292921,0.04616315],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02160893,"threshold_uncertainty_score":0.07228905,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3028657247","doi":"","title":"A Robust Self-Learning Method for Fully Unsupervised Cross-Lingual Mappings of Word Embeddings: Making the Method Robustly Reproducible as Well","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Robustness (evolution); Hyperparameter; Word (group theory); Artificial intelligence; Grid; Stability (learning theory); Unsupervised learning; Machine learning; Natural language processing; Mathematics","authors":[{"name":"Nicolas Garneau","is_ca":true},{"name":"Mathieu Godbout","is_ca":true},{"name":"David Beauchemin","is_ca":true},{"name":"Audrey Durand","is_ca":true},{"name":"Luc Lamontagne","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0626216750026778,"gpt":0.3642289984244982,"spread":0.3016073234218204,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.005924654,0.001793255,0.001492739,0.002367959,0.001316843,0.002602891,0.002875596,0.002534539,0.0043305],"category_scores_gemma":[0.02082757,0.001040994,0.001730147,0.002181029,0.00124739,0.004948051,0.005627324,0.003765027,0.007922208],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000619293,"about_ca_system_score_gemma":0.002576906,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003008452,"about_ca_topic_score_gemma":0.006219768,"domain_scores_codex":[0.9923732,0.002584519,0.0005655446,0.002590795,0.001580301,0.0003056873],"domain_scores_gemma":[0.9880537,0.003321693,0.0004762593,0.004236574,0.003616198,0.0002955626],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003307281,0.0003680361,0.002654289,0.0002867462,0.0004975576,0.0001516967,0.0004079369,0.03572384,0.03514908,0.01230119,0.02151912,0.8906097],"study_design_scores_gemma":[0.00009746222,0.0001721784,0.001701014,0.00004909404,0.0001038266,0.0003962494,0.0001855384,0.9143105,0.04183577,0.02888353,0.01215193,0.0001130166],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.006511368,0.0001263697,0.9863354,0.00009187203,0.0001545062,0.00009120083,0.0002906414,0.005769272,0.000629361],"genre_scores_gemma":[0.1210895,0.0001559901,0.8656151,0.0002608107,0.0001619687,0.0005324552,0.003415916,0.002832916,0.005935428],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9940754,"threshold_uncertainty_score":0.03133297,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3029712483","doi":"","title":"Evaluating the Impact of Sub-word Information and Cross-lingual Word Embeddings on Mi’kmaq Language Modelling","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of New Brunswick","funders":"","keywords":"Word (group theory); Computer science; Language model; Natural language processing; Artificial intelligence; Indigenous language; Linguistics; Indigenous","authors":[{"name":"Jeremie Boudreau","is_ca":false},{"name":"Akankshya Patra","is_ca":false},{"name":"Ashima Suvarna","is_ca":false},{"name":"Paul Cook","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04360119139160799,"gpt":0.3828779568686983,"spread":0.3392767654770903,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008722802,0.002061209,0.001074572,0.00245096,0.0009325017,0.00306363,0.001980967,0.001948529,0.004441273],"category_scores_gemma":[0.03581233,0.000618205,0.001314978,0.002233734,0.0008367283,0.007737929,0.002865872,0.00252665,0.002492253],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001418122,"about_ca_system_score_gemma":0.002024506,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03672968,"about_ca_topic_score_gemma":0.03301183,"domain_scores_codex":[0.9920254,0.004814218,0.0006949856,0.001238303,0.0008478443,0.0003791706],"domain_scores_gemma":[0.9687409,0.02512864,0.0004387133,0.002328484,0.002747961,0.0006152811],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00650753,0.002292757,0.03061942,0.001512379,0.001976655,0.0003973734,0.001064696,0.3347139,0.01257837,0.005013218,0.01389672,0.5894269],"study_design_scores_gemma":[0.0001170418,0.0005223123,0.004014648,0.00007887805,0.000344825,0.000112473,0.000700795,0.9779755,0.01048663,0.003023161,0.002549969,0.00007376164],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.8485186,0.003277872,0.1177187,0.001473098,0.0007099759,0.0003396554,0.005752017,0.0104263,0.01178386],"genre_scores_gemma":[0.9187188,0.0005444863,0.06586937,0.0001927002,0.00005477988,0.000130958,0.01122233,0.0005426561,0.002723952],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.03672968,"threshold_uncertainty_score":0.07303178,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3032504444","doi":"","title":"PhonBank and Data Sharing: Recent Developments in European Portuguese","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Portuguese; Computer science; Data sharing; Data science","authors":[{"name":"Ana Margarida Ramalho","is_ca":false},{"name":"Maria João Freitas","is_ca":false},{"name":"Yvan Rose","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1296680340292582,"gpt":0.3399750691332857,"spread":0.2103070351040275,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.0464945,0.0009511068,0.001410001,0.0112708,0.002868735,0.01072185,0.003666379,0.00230576,0.007652842],"category_scores_gemma":[0.06163608,0.0006788203,0.0008325431,0.02222892,0.004384232,0.01909598,0.007653982,0.001727021,0.002027055],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005165192,"about_ca_system_score_gemma":0.01226062,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0213927,"about_ca_topic_score_gemma":0.01240966,"domain_scores_codex":[0.9744207,0.01025348,0.003189025,0.004038786,0.007047218,0.001050724],"domain_scores_gemma":[0.9431267,0.02833678,0.004047472,0.01155917,0.01036246,0.002567453],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001828702,0.0007066241,0.02789783,0.004440672,0.0003541302,0.0007513377,0.01117286,0.005210933,0.007031627,0.1679335,0.04619628,0.7264754],"study_design_scores_gemma":[0.0002753717,0.0003600128,0.05977973,0.004016492,0.0006083522,0.001338413,0.01307162,0.02074573,0.02014608,0.08400419,0.7953187,0.000335257],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.3373052,0.04050773,0.3156859,0.03234574,0.002291256,0.001763342,0.05358126,0.01389863,0.202621],"genre_scores_gemma":[0.639755,0.01731842,0.225948,0.002528122,0.0008217717,0.001216987,0.08412026,0.005927137,0.02236434],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9963336,"threshold_uncertainty_score":0.2458894,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3030000665","doi":"","title":"FAB: The French Absolute Beginner Corpus for Pronunciation Training.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Phonetics and Phonology Research","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Pronunciation; Context (archaeology); Natural language processing; Set (abstract data type); Partition (number theory); Artificial intelligence; Training set; Speech recognition; Word (group theory); Linguistics; Mathematics","authors":[{"name":"Sean Robertson","is_ca":true},{"name":"Cosmin Munteanu","is_ca":true},{"name":"Gerald Penn","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1083050715093138,"gpt":0.3766665924784349,"spread":0.268361520969121,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001363381,0.00152322,0.0007580734,0.003618473,0.00137883,0.00160403,0.0011813,0.001078228,0.08304247],"category_scores_gemma":[0.005184723,0.0002943287,0.0002939756,0.001912077,0.0006146104,0.001004024,0.001726092,0.001043345,0.04305807],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00122695,"about_ca_system_score_gemma":0.002657938,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.06045381,"about_ca_topic_score_gemma":0.0552521,"domain_scores_codex":[0.9986871,0.0004419007,0.0001083849,0.0002839156,0.0003440563,0.0001346812],"domain_scores_gemma":[0.9966312,0.000965055,0.0001026512,0.0005223898,0.001445872,0.0003328252],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009923768,0.0002524861,0.004910388,0.001500127,0.00005948222,0.0006797765,0.002000235,0.001016875,0.01456059,0.005670175,0.7009569,0.2674007],"study_design_scores_gemma":[0.0002712409,0.0001709102,0.04984272,0.0003852732,0.00004576562,0.001411524,0.001386533,0.001511286,0.01079948,0.001953895,0.9321259,0.00009553322],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.05868372,0.003757578,0.01437077,0.001186257,0.0007280223,0.0005631123,0.7990407,0.01751536,0.1041544],"genre_scores_gemma":[0.1647836,0.0008694881,0.01959682,0.0004153172,0.00024306,0.001133592,0.7637794,0.004505772,0.04467303],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.08304247,"threshold_uncertainty_score":0.2778047,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3029614628","doi":"","title":"CantoMap: a Hong Kong Cantonese MapTask Corpus.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Geographic Information Systems Studies","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Transcription (linguistics); Computer science; Utterance; Annotation; Natural language processing; Phonetic transcription; Phonology; Artificial intelligence; Task (project management); Speech recognition; Linguistics; Engineering","authors":[{"name":"Grégoire Winterstein","is_ca":true},{"name":"Carmen Tang","is_ca":false},{"name":"Regine Lai","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03570656174926953,"gpt":0.3177618021598226,"spread":0.2820552404105531,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001391575,0.001425833,0.0006555517,0.003539092,0.002434782,0.001963451,0.001576084,0.0008580184,0.0275677],"category_scores_gemma":[0.004974036,0.0003561121,0.0003428818,0.005104829,0.0008062538,0.001399556,0.002854362,0.0009260446,0.01244512],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002219176,"about_ca_system_score_gemma":0.005622915,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.3142273,"about_ca_topic_score_gemma":0.3341503,"domain_scores_codex":[0.9991183,0.0002756794,0.0001224205,0.000185925,0.0001741074,0.0001235189],"domain_scores_gemma":[0.9968488,0.0008655189,0.0001343857,0.0006106595,0.001190614,0.0003500379],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.0006668362,0.0001830906,0.01219819,0.002952089,0.0001335012,0.001272481,0.004516386,0.001009434,0.007871588,0.003547505,0.9064352,0.05921372],"study_design_scores_gemma":[0.0003205184,0.000080318,0.1334359,0.0006380258,0.0001746118,0.000729436,0.007368909,0.002954089,0.005509465,0.001372629,0.8472763,0.0001397836],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.06250612,0.0008655821,0.002501471,0.000568067,0.0002463208,0.0005237781,0.9047052,0.001624075,0.02645947],"genre_scores_gemma":[0.05528621,0.0002326421,0.003505838,0.00009381626,0.00002723094,0.001137218,0.9311336,0.0003781524,0.008205255],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3142273,"threshold_uncertainty_score":0.6247966,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}