{"meta":{"query_hash":"84f0ed203413","filters":{"venue":"Language Resources and Evaluation"},"cohort_total":41,"direct_labels_cover":0,"predictions_cover":41,"exported":41,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/84f0ed203413","api":"https://metacan.xera.ac/api/v1/cohort?venue=Language+Resources+and+Evaluation"},"results":[{"id":"W1921587136","doi":"10.1007/s10579-015-9318-3","title":"Cross level semantic similarity: an evaluation framework for universal measures of similarity","year":2015,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"European Research Council","keywords":"Computer science; Natural language processing; Similarity (geometry); Semantic similarity; Sentence; Task (project management); Artificial intelligence; Word (group theory); SemEval; Paragraph; Process (computing); Meaning (existential); WordNet; Information retrieval; Linguistics; Psychology; World Wide Web","score_opus":0.21232820109561595,"score_gpt":0.3869887300338786,"score_spread":0.17466052893826267,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1921587136","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.021688528,0.0010031083,0.97204685,0.00014308491,0.000088809225,0.0004102549,0.0006332746,0.0013459966,0.0026400108],"genre_scores_gemma":[0.40879327,0.00055454083,0.58532137,0.00014636011,0.00019346453,0.0011357986,0.0019179165,0.0005842705,0.0013531148],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97493875,0.01090835,0.0027606743,0.002648959,0.008026661,0.0007165587],"domain_scores_gemma":[0.9645081,0.019548161,0.0023981351,0.0063810423,0.0059709023,0.0011935553],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024129214,0.0013911831,0.0020710602,0.012123618,0.0014304467,0.0053690686,0.0024734442,0.0021485167,0.0029954636],"category_scores_gemma":[0.06528036,0.00052313606,0.001974848,0.007821052,0.0019568354,0.011363664,0.005723787,0.002126443,0.00087291974],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019071632,0.00091230817,0.028248578,0.0016784037,0.0017763411,0.00023768007,0.0026437575,0.028456664,0.019701356,0.18722643,0.010013435,0.7171978],"study_design_scores_gemma":[0.00020196753,0.0020268126,0.022483358,0.00056112825,0.001246642,0.0011793971,0.0015848479,0.5581739,0.029200079,0.3625684,0.020391693,0.00038172404],"about_ca_topic_score_codex":0.001939076,"about_ca_topic_score_gemma":0.0018627377,"teacher_disagreement_score":0.024129214,"about_ca_system_score_codex":0.0019513459,"about_ca_system_score_gemma":0.0019573811,"threshold_uncertainty_score":0.12760901},"labels":[],"label_agreement":null},{"id":"W1990825561","doi":"10.1007/s10579-008-9072-x","title":"Disambiguation of partial cognates","year":2008,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Cognate; Computer science; Natural language processing; Meaning (existential); Context (archaeology); Artificial intelligence; Word (group theory); Word-sense disambiguation; Machine translation; Linguistics; Psychology; Biology","score_opus":0.024028377565879798,"score_gpt":0.3017247175024333,"score_spread":0.2776963399365535,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1990825561","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6149464,0.00576524,0.31771243,0.0014550672,0.0009577194,0.0006895741,0.0048763915,0.012086456,0.041510735],"genre_scores_gemma":[0.8598629,0.0007685002,0.12552702,0.0002803565,0.00018952382,0.0001334095,0.005487444,0.0012066874,0.0065439995],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9912361,0.0036727358,0.0010356903,0.0017416734,0.0016690759,0.0006447629],"domain_scores_gemma":[0.98567617,0.0072215837,0.0004057398,0.002683434,0.003360565,0.0006524947],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061160387,0.0014597352,0.002303403,0.005921434,0.003163668,0.005768831,0.0020420318,0.0018114211,0.008848233],"category_scores_gemma":[0.021465054,0.00069514266,0.0015359846,0.0024295684,0.0017877342,0.010530666,0.005356506,0.0017131853,0.0030974403],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004896315,0.00060613337,0.019059723,0.0016155575,0.00066002546,0.0016622196,0.0024975943,0.017934924,0.050799996,0.06224063,0.021059701,0.8169671],"study_design_scores_gemma":[0.0005376345,0.0009701094,0.015281488,0.0005511554,0.0018818927,0.0050650216,0.0053892117,0.4316488,0.25222245,0.21089761,0.07501578,0.0005388839],"about_ca_topic_score_codex":0.0045514232,"about_ca_topic_score_gemma":0.0056298412,"teacher_disagreement_score":0.008848233,"about_ca_system_score_codex":0.0010176038,"about_ca_system_score_gemma":0.0028566916,"threshold_uncertainty_score":0.032345057},"labels":[],"label_agreement":null},{"id":"W2037256905","doi":"10.1007/s10579-014-9271-6","title":"A qualitative comparison method for rhetorical structures: identifying different discourse structures in multilingual corpora","year":2014,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":81,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Rhetorical question; Linguistics; Computer science; Natural language processing; Annotation; Translation (biology); Artificial intelligence; Contrastive linguistics; Applied linguistics; Philosophy","score_opus":0.08234689324121308,"score_gpt":0.47851651750124213,"score_spread":0.39616962426002905,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2037256905","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11223794,0.00078777754,0.81558204,0.0011972412,0.00031829166,0.018403498,0.009043007,0.0012204258,0.04120978],"genre_scores_gemma":[0.2523417,0.0002599571,0.70554495,0.0003151655,0.00004421625,0.033671506,0.0027975196,0.0003893218,0.004635634],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.94543654,0.03518339,0.00460223,0.0045378846,0.009237705,0.0010022392],"domain_scores_gemma":[0.8347893,0.109949395,0.006716553,0.008818467,0.038084365,0.0016419892],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.044161096,0.000992284,0.0011659275,0.013162892,0.0041945227,0.004793654,0.0025422813,0.0012595208,0.011526145],"category_scores_gemma":[0.13460061,0.00070950773,0.0011270446,0.01009224,0.0040522013,0.003713692,0.0048719165,0.0014919321,0.0015162635],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030170546,0.0012662899,0.028478863,0.010443087,0.00059590366,0.00052169367,0.15613385,0.0024991827,0.051448576,0.14783335,0.018912002,0.57885015],"study_design_scores_gemma":[0.0022694399,0.0024644667,0.07732867,0.0052656196,0.0015530172,0.0016223148,0.23636182,0.052617572,0.13563798,0.2517841,0.23204972,0.0010453789],"about_ca_topic_score_codex":0.0046013403,"about_ca_topic_score_gemma":0.009983829,"teacher_disagreement_score":0.044161096,"about_ca_system_score_codex":0.0056877863,"about_ca_system_score_gemma":0.007736226,"threshold_uncertainty_score":0.233549},"labels":[],"label_agreement":null},{"id":"W2062046281","doi":"10.1007/s10579-009-9083-2","title":"Classification of semantic relations between nominals","year":2009,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":63,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada; University of Ottawa","funders":"","keywords":"Computer science; Task (project management); Natural language processing; SemEval; Sentence; Artificial intelligence; Process (computing); Semantic similarity; Linguistics","score_opus":0.030966356234334785,"score_gpt":0.33024747582626307,"score_spread":0.29928111959192827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2062046281","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7705426,0.0027312594,0.17110908,0.0016832642,0.00039601838,0.0007887511,0.011246281,0.0040014884,0.037501216],"genre_scores_gemma":[0.9145765,0.0005040888,0.06927261,0.00008638743,0.00009531044,0.00016966376,0.011411955,0.00022259595,0.0036608912],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99486536,0.00153901,0.00056599843,0.00079688133,0.0018820871,0.00035071425],"domain_scores_gemma":[0.9796158,0.0120033035,0.0013066238,0.0017166453,0.004554663,0.0008029657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048829443,0.0005685298,0.0007210122,0.007751164,0.001641432,0.0037380795,0.0014066899,0.0011872284,0.0059330375],"category_scores_gemma":[0.022168873,0.00021903966,0.0008952202,0.0033600926,0.0010424724,0.0060146237,0.00175972,0.001125461,0.0016510535],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0036588442,0.0010713787,0.10042476,0.0011622194,0.00023190262,0.00058246404,0.0019509936,0.008358469,0.031434085,0.06389728,0.020121392,0.7671063],"study_design_scores_gemma":[0.00032849438,0.0010272036,0.13452798,0.0006215729,0.0008854677,0.001487108,0.008238938,0.55732405,0.083345555,0.14700766,0.06497576,0.00023021745],"about_ca_topic_score_codex":0.0069790664,"about_ca_topic_score_gemma":0.0071464963,"teacher_disagreement_score":0.007751164,"about_ca_system_score_codex":0.0019897625,"about_ca_system_score_gemma":0.0022175733,"threshold_uncertainty_score":0.025823772},"labels":[],"label_agreement":null},{"id":"W2250536124","doi":"","title":"Discovering frames in specialized domains","year":2014,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"FrameNet; Computer science; Complement (music); Field (mathematics); Artificial intelligence; Natural language processing; Information retrieval; Parsing; Mathematics","score_opus":0.011399440839449463,"score_gpt":0.2946918213716688,"score_spread":0.2832923805322194,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2250536124","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41011974,0.0020587407,0.55894136,0.00146152,0.00015509484,0.0006675455,0.005838694,0.0066876737,0.014069563],"genre_scores_gemma":[0.70848316,0.00092716084,0.2723902,0.00024068086,0.00010541125,0.00019827289,0.012108729,0.0006971947,0.004849269],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9961202,0.0012581941,0.00028523305,0.0010352691,0.00081068795,0.00049039867],"domain_scores_gemma":[0.9907265,0.0055450033,0.00043171743,0.0015315334,0.0013207643,0.00044454992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026232232,0.0013346843,0.0015756583,0.0066907303,0.0016889048,0.0038236028,0.002301669,0.0020316483,0.007951257],"category_scores_gemma":[0.014247846,0.00070108986,0.001619433,0.003815797,0.0011489973,0.00955022,0.0029345094,0.0018331743,0.0022174825],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002561462,0.0012876819,0.031422596,0.001002252,0.00045900972,0.0018488746,0.0021206883,0.032114815,0.030841712,0.07739259,0.023610396,0.7953379],"study_design_scores_gemma":[0.0002988354,0.0005231222,0.009576034,0.00029668122,0.00070761156,0.00082268304,0.0042038662,0.7242944,0.058071848,0.17430757,0.02678604,0.000111278074],"about_ca_topic_score_codex":0.013325788,"about_ca_topic_score_gemma":0.020223152,"teacher_disagreement_score":0.013325788,"about_ca_system_score_codex":0.001808833,"about_ca_system_score_gemma":0.0026336159,"threshold_uncertainty_score":0.026599586},"labels":[],"label_agreement":null},{"id":"W2251445713","doi":"","title":"Capturing syntactico-semantic regularities among terms: An application of the FrameNet methodology to terminology","year":2012,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"FrameNet; Computer science; Terminology; Annotation; Focus (optics); Natural language processing; Artificial intelligence; Point (geometry); Resource (disambiguation); Lexicon; Computational linguistics; Semantics (computer science); Linguistics; Parsing; Programming language","score_opus":0.03784322549439183,"score_gpt":0.3385109969376508,"score_spread":0.30066777144325896,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2251445713","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03556982,0.0005616429,0.9552153,0.00063761347,0.00006405993,0.0003107327,0.0013795337,0.0008436173,0.0054177474],"genre_scores_gemma":[0.34298027,0.0006445834,0.6509994,0.00012266416,0.00009275464,0.0003823841,0.0028016346,0.0004037556,0.0015724885],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99304855,0.0030874289,0.0007766063,0.0010150187,0.0017360752,0.00033637224],"domain_scores_gemma":[0.9868073,0.007427548,0.001155709,0.0019125951,0.0023856855,0.00031126765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008608997,0.00093339523,0.0016129145,0.009330446,0.0019354278,0.0047161463,0.0021974163,0.0012935319,0.0030477028],"category_scores_gemma":[0.02592189,0.00068961387,0.001542614,0.007895039,0.0025541587,0.013722197,0.003695484,0.0016695061,0.0005761269],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037805812,0.00021276136,0.009640882,0.0007437462,0.00020574783,0.00044882175,0.003538319,0.02434735,0.011048693,0.6117911,0.0048611583,0.3327834],"study_design_scores_gemma":[0.000039172377,0.00010628677,0.0035336635,0.00027346742,0.00028787734,0.00040402415,0.0016770195,0.2925686,0.011033356,0.6658524,0.024118556,0.00010560426],"about_ca_topic_score_codex":0.013845355,"about_ca_topic_score_gemma":0.016418558,"teacher_disagreement_score":0.013845355,"about_ca_system_score_codex":0.0027597593,"about_ca_system_score_gemma":0.0033064743,"threshold_uncertainty_score":0.045529246},"labels":[],"label_agreement":null},{"id":"W2577349305","doi":"","title":"Training & Quality Assessment of an Optical Character Recognition Model for Northern Haida.","year":2016,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Handwritten Text Recognition Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Optical character recognition; Computer science; Character (mathematics); Language model; Unicode; Porting; Artificial intelligence; Natural language processing; Hidden Markov model; Speech recognition; Set (abstract data type); Image (mathematics)","score_opus":0.12266473322717336,"score_gpt":0.38797495915431435,"score_spread":0.265310225927141,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2577349305","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9339243,0.0005377874,0.053514384,0.00024126991,0.00018244864,0.00028003711,0.0012911799,0.0047645983,0.00526398],"genre_scores_gemma":[0.961786,0.000119606484,0.027179675,0.00005817198,0.000014173643,0.00008391437,0.0030114024,0.0002189084,0.0075280177],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99942243,0.00010295054,0.00004370331,0.00019583848,0.00017180224,0.00006333779],"domain_scores_gemma":[0.9979572,0.0004228585,0.000089953326,0.00025197282,0.0011792986,0.00009867576],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013803896,0.000733847,0.00046362553,0.0004987013,0.0006773091,0.00081755914,0.0009798482,0.00066148594,0.002599166],"category_scores_gemma":[0.0036782923,0.00032215935,0.0004079424,0.00041747076,0.0003164695,0.0008630855,0.0005934858,0.00052870915,0.0014370942],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017377911,0.0006942674,0.043804914,0.00029167708,0.0002456561,0.0003560559,0.00044704843,0.18791756,0.07649422,0.00047418516,0.012721781,0.6748149],"study_design_scores_gemma":[0.0000626551,0.00038284494,0.037113022,0.000024074416,0.00010016381,0.00008860001,0.00031642764,0.9130394,0.04581007,0.00017945908,0.0028523947,0.00003086968],"about_ca_topic_score_codex":0.14022829,"about_ca_topic_score_gemma":0.1447897,"teacher_disagreement_score":0.14022829,"about_ca_system_score_codex":0.0013062628,"about_ca_system_score_gemma":0.0019486072,"threshold_uncertainty_score":0.2788241},"labels":[],"label_agreement":null},{"id":"W2582792602","doi":"","title":"Detecting semantic changes in Alzheimer’s disease with vector space models","year":2016,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":13,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Disease; Space (punctuation); Medicine; Pathology","score_opus":0.0302507815983099,"score_gpt":0.2936973496375037,"score_spread":0.2634465680391938,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2582792602","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7153819,0.0032037639,0.26839274,0.0012062953,0.00016869669,0.00029479826,0.0054834667,0.003967141,0.0019011438],"genre_scores_gemma":[0.9340327,0.0005317718,0.059358787,0.0001326485,0.000044480184,0.00012044726,0.0048758793,0.00008048608,0.0008228886],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983979,0.00060839986,0.00018869713,0.0003024856,0.00036901986,0.00013345151],"domain_scores_gemma":[0.9959372,0.003069145,0.00021443273,0.00018298041,0.00049958867,0.00009658597],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0030420132,0.00093708653,0.00084036146,0.004642628,0.00046324386,0.0015813796,0.00081289164,0.000952925,0.0009940042],"category_scores_gemma":[0.0072942,0.00017251167,0.0012886757,0.0024171579,0.00040526132,0.001989576,0.0008062003,0.00073341385,0.00035958356],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028041478,0.001222556,0.0711288,0.0005906096,0.0010547368,0.00057423953,0.00035398948,0.21148048,0.008699782,0.005788444,0.009770092,0.686532],"study_design_scores_gemma":[0.000046574227,0.00022150579,0.005412978,0.000031142124,0.00017280345,0.00017062208,0.00017421243,0.9824923,0.0034757112,0.007036544,0.0007468592,0.000018684179],"about_ca_topic_score_codex":0.020351091,"about_ca_topic_score_gemma":0.012166571,"teacher_disagreement_score":0.020351091,"about_ca_system_score_codex":0.0011366252,"about_ca_system_score_gemma":0.001154074,"threshold_uncertainty_score":0.040465236},"labels":[],"label_agreement":null},{"id":"W2583305969","doi":"10.1007/s10579-017-9383-x","title":"RST Signalling Corpus: a corpus of signals of coherence relations","year":2017,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Treebank; Annotation; Corpus linguistics; Natural language processing; Computer science; Parsing; Coherence (philosophical gambling strategy); Artificial intelligence; Text corpus; Linguistics; British National Corpus","score_opus":0.030780822958553684,"score_gpt":0.3193694584320568,"score_spread":0.2885886354735031,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2583305969","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2560739,0.0047461,0.088699825,0.0033437535,0.0013551705,0.0016675736,0.5526342,0.0136403935,0.077839166],"genre_scores_gemma":[0.40793693,0.0016225931,0.0760173,0.00072124874,0.00053736253,0.0024977243,0.47244325,0.0026436914,0.035579912],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99725085,0.0008702228,0.00033101355,0.00046560305,0.0009038898,0.0001785056],"domain_scores_gemma":[0.9850116,0.009570358,0.00072192337,0.0021341804,0.0021164608,0.00044551288],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020753627,0.0011649855,0.00080342835,0.0042862515,0.0015305846,0.0020724798,0.0016552615,0.0025530183,0.030605486],"category_scores_gemma":[0.017156677,0.00065667316,0.00048622757,0.0035712216,0.0013082904,0.0021690559,0.0021013923,0.0017616844,0.014354473],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0039048449,0.0009018078,0.00634534,0.008438109,0.00032413943,0.0026162004,0.00806343,0.005638235,0.14779799,0.046045136,0.4653592,0.3045656],"study_design_scores_gemma":[0.0011084436,0.0005845534,0.0727355,0.00085175893,0.00048194465,0.0042673936,0.0038040902,0.018480184,0.07631247,0.014803617,0.8060129,0.0005571572],"about_ca_topic_score_codex":0.0093255155,"about_ca_topic_score_gemma":0.010638906,"teacher_disagreement_score":0.030605486,"about_ca_system_score_codex":0.0010449528,"about_ca_system_score_gemma":0.0023501462,"threshold_uncertainty_score":0.10238558},"labels":[],"label_agreement":null},{"id":"W2896826757","doi":"10.1007/s10579-018-9430-2","title":"VERTa: a linguistic approach to automatic machine translation evaluation","year":2018,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Ministerio de Asuntos Económicos y Transformación Digital, Gobierno de España; Alberta-Pacific Forest Industries","keywords":"Metric (unit); Computer science; Machine translation; Natural language processing; Variety (cybernetics); Artificial intelligence; Linguistics; Engineering","score_opus":0.030387430137635726,"score_gpt":0.32967932798055893,"score_spread":0.2992918978429232,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2896826757","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.022251097,0.0014121534,0.8927895,0.0006326675,0.0006652174,0.0015060542,0.00542806,0.058305547,0.0170097],"genre_scores_gemma":[0.13944376,0.0006178573,0.8248838,0.0004997357,0.00031803295,0.0016725303,0.014484224,0.0074338075,0.010646231],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.968594,0.020351551,0.002028133,0.0019133639,0.006417021,0.0006959033],"domain_scores_gemma":[0.9791027,0.009897619,0.00076051516,0.003166916,0.00660485,0.00046746773],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013276025,0.0022322086,0.0021224096,0.008677003,0.0023646296,0.0067567336,0.0036784755,0.0023178966,0.010808187],"category_scores_gemma":[0.030831141,0.0012158672,0.0015460697,0.0046509625,0.0013937664,0.006629603,0.005087656,0.0029534765,0.0055472804],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013630812,0.0010324456,0.0029721237,0.0019975563,0.0009867551,0.00048035118,0.0011558372,0.017858395,0.044398576,0.04210081,0.10496585,0.7806882],"study_design_scores_gemma":[0.0006484146,0.0011830562,0.0039980654,0.00047647796,0.00072166027,0.0010569688,0.0011184325,0.7118655,0.08275644,0.07799259,0.117762595,0.00041984738],"about_ca_topic_score_codex":0.004772209,"about_ca_topic_score_gemma":0.008031053,"teacher_disagreement_score":0.013276025,"about_ca_system_score_codex":0.0015547674,"about_ca_system_score_gemma":0.0035587277,"threshold_uncertainty_score":0.07021117},"labels":[],"label_agreement":null},{"id":"W3028657247","doi":"","title":"A Robust Self-Learning Method for Fully Unsupervised Cross-Lingual Mappings of Word Embeddings: Making the Method Robustly Reproducible as Well","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Computer science; Robustness (evolution); Hyperparameter; Word (group theory); Artificial intelligence; Grid; Stability (learning theory); Unsupervised learning; Machine learning; Natural language processing; Mathematics","score_opus":0.0626216750026778,"score_gpt":0.3642289984244982,"score_spread":0.3016073234218204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3028657247","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006511368,0.00012636973,0.9863354,0.00009187203,0.00015450618,0.000091200825,0.00029064136,0.0057692723,0.000629361],"genre_scores_gemma":[0.12108947,0.00015599014,0.86561507,0.00026081066,0.00016196865,0.00053245516,0.0034159156,0.002832916,0.005935428],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99237317,0.0025845193,0.00056554464,0.0025907948,0.0015803009,0.00030568725],"domain_scores_gemma":[0.98805374,0.0033216928,0.00047625927,0.004236574,0.0036161977,0.00029556258],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.005924654,0.0017932549,0.0014927393,0.0023679594,0.0013168431,0.0026028906,0.0028755958,0.0025345387,0.0043305005],"category_scores_gemma":[0.020827573,0.0010409942,0.001730147,0.0021810294,0.0012473898,0.004948051,0.005627324,0.0037650273,0.007922208],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003307281,0.00036803607,0.0026542887,0.00028674616,0.00049755763,0.00015169672,0.0004079369,0.03572384,0.03514908,0.0123011945,0.021519119,0.89060974],"study_design_scores_gemma":[0.00009746222,0.00017217836,0.0017010135,0.000049094044,0.000103826584,0.00039624935,0.00018553837,0.91431046,0.041835774,0.02888353,0.0121519305,0.00011301664],"about_ca_topic_score_codex":0.003008452,"about_ca_topic_score_gemma":0.0062197684,"teacher_disagreement_score":0.99407536,"about_ca_system_score_codex":0.000619293,"about_ca_system_score_gemma":0.0025769058,"threshold_uncertainty_score":0.03133297},"labels":[],"label_agreement":null},{"id":"W3028807070","doi":"","title":"Extraction of Hyponymic Relations in French with Knowledge-Pattern-Based Word Sketches.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec","funders":"","keywords":"Computer science; Sketch; Natural language processing; Artificial intelligence; Word (group theory); Grammar; Domain (mathematical analysis); Thesaurus; Process (computing); Information extraction; Information retrieval; Linguistics","score_opus":0.021142584555562678,"score_gpt":0.29778985129183766,"score_spread":0.276647266736275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3028807070","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5018258,0.0060219257,0.38191748,0.0010832351,0.00033221475,0.0012669562,0.054479968,0.029933337,0.023139047],"genre_scores_gemma":[0.6395427,0.0014269805,0.29002005,0.00013513844,0.00008488107,0.0004024233,0.06036721,0.0009240129,0.0070966356],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99915814,0.00024491342,0.00010480086,0.0002825802,0.00015322151,0.00005640501],"domain_scores_gemma":[0.9971107,0.0017942967,0.00018238123,0.0002181132,0.00060175225,0.000092837625],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00069470255,0.000888418,0.0005514081,0.005005981,0.0006419041,0.0016604806,0.0006493345,0.00077441445,0.008609612],"category_scores_gemma":[0.005581913,0.00027499738,0.00076620036,0.0022498458,0.00038437283,0.0027433606,0.0010277281,0.00064371014,0.003558904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009723075,0.0002692672,0.016392764,0.003049782,0.00025781724,0.0016923952,0.0031332045,0.005318453,0.0839956,0.014972135,0.025103947,0.84484226],"study_design_scores_gemma":[0.00062926527,0.0012117832,0.102196455,0.0010513104,0.0011961006,0.0066041816,0.012504187,0.31684932,0.1583902,0.045153033,0.35389614,0.0003180351],"about_ca_topic_score_codex":0.016626177,"about_ca_topic_score_gemma":0.018622916,"teacher_disagreement_score":0.016626177,"about_ca_system_score_codex":0.00063694536,"about_ca_system_score_gemma":0.0013696405,"threshold_uncertainty_score":0.033058822},"labels":[],"label_agreement":null},{"id":"W3029096252","doi":"","title":"A Lexicon-Based Approach for Detecting Hedges in Informal Text","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Computer science; Hedge; Lexicon; Natural language processing; Sentence; Interview; Artificial intelligence; Part-of-speech tagging; Linguistics; Part of speech; Sociology","score_opus":0.03137950489801756,"score_gpt":0.300703666162043,"score_spread":0.2693241612640255,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029096252","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11037575,0.001223971,0.8496416,0.0006671164,0.00016698401,0.00177843,0.0057747,0.0191111,0.011260258],"genre_scores_gemma":[0.41507038,0.0003425179,0.5716984,0.00022316001,0.000085310756,0.00060676195,0.0073906113,0.0006437546,0.0039391285],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9951448,0.001625462,0.00066842936,0.00062347145,0.0016997816,0.00023801648],"domain_scores_gemma":[0.98874485,0.005880194,0.00075614476,0.001021963,0.0031756382,0.00042130108],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029357919,0.001012575,0.001150095,0.010203019,0.0012591209,0.004010528,0.0014451741,0.0016272693,0.0044845175],"category_scores_gemma":[0.013861601,0.0005037769,0.00081378984,0.003971232,0.00088621775,0.0047040638,0.0023530438,0.0011033057,0.0022646992],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00081861916,0.00080502656,0.018933747,0.0010717184,0.00030420133,0.0010508591,0.0018760443,0.0060852016,0.09664157,0.023047073,0.022099229,0.82726675],"study_design_scores_gemma":[0.00036255628,0.0010322506,0.026328977,0.00042107995,0.0007855103,0.0025809149,0.003098581,0.75227845,0.10982754,0.05682352,0.046051793,0.00040883425],"about_ca_topic_score_codex":0.006771499,"about_ca_topic_score_gemma":0.011549169,"teacher_disagreement_score":0.010203019,"about_ca_system_score_codex":0.0011098805,"about_ca_system_score_gemma":0.0023658145,"threshold_uncertainty_score":0.0155261755},"labels":[],"label_agreement":null},{"id":"W3029265130","doi":"","title":"Multilingual Dictionary Based Construction of Core Vocabulary.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Vocabulary; Natural language processing; Core (optical fiber); Artificial intelligence; Set (abstract data type); Bilingual dictionary; Field (mathematics); Resource (disambiguation); Machine translation; Linguistics; Programming language","score_opus":0.028694671351722077,"score_gpt":0.3027328626012327,"score_spread":0.2740381912495106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029265130","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15100893,0.0016002014,0.7510712,0.00055596064,0.0004962291,0.0023552836,0.021600045,0.01650844,0.054803614],"genre_scores_gemma":[0.47190988,0.0006737193,0.46175256,0.00020992028,0.00007120408,0.0012733493,0.047321428,0.0023178635,0.014470148],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.995415,0.0019576054,0.0006137853,0.00081349374,0.0009146866,0.00028546556],"domain_scores_gemma":[0.99194896,0.0024788082,0.00026219522,0.0011297462,0.0038340536,0.00034628587],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026593744,0.0006668834,0.00089886336,0.004443281,0.0011197777,0.0027894143,0.0013507651,0.0006210512,0.014485645],"category_scores_gemma":[0.013567471,0.00042233735,0.00063853327,0.0026290007,0.000666392,0.007674701,0.0046506734,0.0012083593,0.0068703895],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010801076,0.00060021254,0.00880594,0.0020041289,0.00022163775,0.00038884688,0.002459609,0.004986783,0.061078295,0.058575865,0.0412084,0.8185901],"study_design_scores_gemma":[0.0006009321,0.001316769,0.016810883,0.0010609475,0.00084714266,0.0027819118,0.010712795,0.30541906,0.26967573,0.10784843,0.28259328,0.00033204],"about_ca_topic_score_codex":0.007330266,"about_ca_topic_score_gemma":0.012024214,"teacher_disagreement_score":0.014485645,"about_ca_system_score_codex":0.0011711597,"about_ca_system_score_gemma":0.003453089,"threshold_uncertainty_score":0.04845935},"labels":[],"label_agreement":null},{"id":"W3029614628","doi":"","title":"CantoMap: a Hong Kong Cantonese MapTask Corpus.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Geographic Information Systems Studies","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Transcription (linguistics); Computer science; Utterance; Annotation; Natural language processing; Phonetic transcription; Phonology; Artificial intelligence; Task (project management); Speech recognition; Linguistics; Engineering","score_opus":0.03570656174926953,"score_gpt":0.31776180215982264,"score_spread":0.2820552404105531,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029614628","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06250612,0.00086558214,0.0025014714,0.00056806696,0.00024632076,0.00052377814,0.90470517,0.0016240745,0.02645947],"genre_scores_gemma":[0.055286206,0.0002326421,0.003505838,0.000093816256,0.000027230943,0.0011372183,0.9311336,0.00037815236,0.008205255],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.99911827,0.0002756794,0.00012242046,0.000185925,0.00017410742,0.00012351894],"domain_scores_gemma":[0.99684876,0.0008655189,0.00013438571,0.0006106595,0.0011906141,0.0003500379],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013915753,0.0014258329,0.0006555517,0.0035390917,0.0024347815,0.001963451,0.0015760837,0.0008580184,0.027567696],"category_scores_gemma":[0.004974036,0.00035611205,0.0003428818,0.005104829,0.00080625375,0.0013995563,0.0028543617,0.0009260446,0.012445122],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006668362,0.0001830906,0.012198191,0.0029520888,0.00013350123,0.0012724808,0.004516386,0.0010094338,0.007871588,0.003547505,0.9064352,0.059213724],"study_design_scores_gemma":[0.00032051842,0.000080318,0.13343595,0.00063802575,0.00017461176,0.000729436,0.0073689087,0.002954089,0.0055094655,0.0013726286,0.8472763,0.00013978362],"about_ca_topic_score_codex":0.31422734,"about_ca_topic_score_gemma":0.33415034,"teacher_disagreement_score":0.31422734,"about_ca_system_score_codex":0.0022191764,"about_ca_system_score_gemma":0.0056229155,"threshold_uncertainty_score":0.6247966},"labels":[],"label_agreement":null},{"id":"W3029712483","doi":"","title":"Evaluating the Impact of Sub-word Information and Cross-lingual Word Embeddings on Mi’kmaq Language Modelling","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Word (group theory); Computer science; Language model; Natural language processing; Artificial intelligence; Indigenous language; Linguistics; Indigenous","score_opus":0.04360119139160799,"score_gpt":0.3828779568686983,"score_spread":0.3392767654770903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029712483","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84851855,0.0032778725,0.11771875,0.0014730984,0.0007099759,0.0003396554,0.005752017,0.010426304,0.011783855],"genre_scores_gemma":[0.91871876,0.0005444863,0.06586937,0.00019270017,0.00005477988,0.00013095797,0.011222334,0.0005426561,0.002723952],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99202543,0.0048142183,0.0006949856,0.0012383033,0.0008478443,0.00037917058],"domain_scores_gemma":[0.96874094,0.025128644,0.00043871326,0.0023284839,0.0027479606,0.00061528105],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008722802,0.002061209,0.0010745724,0.00245096,0.00093250175,0.0030636296,0.0019809674,0.0019485293,0.004441273],"category_scores_gemma":[0.035812333,0.000618205,0.0013149782,0.0022337344,0.00083672826,0.0077379285,0.002865872,0.0025266502,0.0024922525],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00650753,0.0022927572,0.03061942,0.0015123787,0.0019766546,0.00039737337,0.0010646958,0.33471388,0.01257837,0.0050132177,0.013896716,0.58942693],"study_design_scores_gemma":[0.00011704185,0.0005223123,0.0040146476,0.00007887805,0.00034482498,0.00011247295,0.00070079503,0.97797555,0.010486633,0.0030231613,0.0025499694,0.000073761636],"about_ca_topic_score_codex":0.03672968,"about_ca_topic_score_gemma":0.03301183,"teacher_disagreement_score":0.03672968,"about_ca_system_score_codex":0.0014181222,"about_ca_system_score_gemma":0.0020245055,"threshold_uncertainty_score":0.07303178},"labels":[],"label_agreement":null},{"id":"W3029889931","doi":"","title":"The Johns Hopkins University Bible Corpus: 1600+ Tongues for Typological Exploration","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Natural language processing; Variety (cybernetics); Representation (politics); Linguistics; Corpus linguistics; Parallel corpora; Pronoun; Artificial intelligence; Information retrieval; History; Machine translation; Philosophy","score_opus":0.04572153790965464,"score_gpt":0.29345114168665953,"score_spread":0.2477296037770049,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029889931","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20184046,0.0035190403,0.0070909476,0.0016580212,0.0012886835,0.00056106,0.6098959,0.001563343,0.17258255],"genre_scores_gemma":[0.2664446,0.0020319698,0.018978352,0.0005498966,0.00056886167,0.0014424484,0.64400786,0.0017550401,0.064220935],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.998987,0.000280591,0.00015222114,0.00013518405,0.0003599264,0.00008511085],"domain_scores_gemma":[0.9962239,0.0012925752,0.0002337776,0.00065624976,0.0012055004,0.00038785482],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010440353,0.00050288433,0.00047750966,0.009435523,0.0024899822,0.0018894847,0.00072999985,0.00062356837,0.053078637],"category_scores_gemma":[0.006478014,0.00028087696,0.00016295395,0.010030854,0.0011041999,0.0011863298,0.0023655738,0.00082779356,0.025947822],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063549454,0.00020049044,0.0079096,0.002174381,0.00003442351,0.0010203221,0.011500407,0.00048402665,0.013147167,0.01575643,0.75519085,0.19194643],"study_design_scores_gemma":[0.00010772734,0.000057691548,0.052345827,0.00051518134,0.00003638779,0.00066208845,0.0057077613,0.0007425279,0.005066743,0.001955146,0.9327561,0.00004677055],"about_ca_topic_score_codex":0.014629666,"about_ca_topic_score_gemma":0.030219562,"teacher_disagreement_score":0.053078637,"about_ca_system_score_codex":0.0009811205,"about_ca_system_score_gemma":0.0026501743,"threshold_uncertainty_score":0.17756575},"labels":[],"label_agreement":null},{"id":"W3029927342","doi":"","title":"Contextualized Embeddings based Transformer Encoder for Sentence Similarity Modeling in Answer Selection Task","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":54,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Transformer; Encoder; Computer science; Sentence; Artificial intelligence; Language model; Natural language processing; Feature selection; Selection (genetic algorithm); Engineering; Voltage","score_opus":0.057425820451554886,"score_gpt":0.309027963334743,"score_spread":0.25160214288318816,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029927342","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16390093,0.0016445526,0.80471325,0.000653683,0.00066853873,0.00041708746,0.0055663637,0.016658487,0.005777076],"genre_scores_gemma":[0.7925718,0.0005342071,0.1862885,0.00023800074,0.000245496,0.00037974893,0.013337099,0.00058804377,0.005817258],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99887747,0.000384022,0.0000871991,0.00029986975,0.00021160273,0.000139867],"domain_scores_gemma":[0.99816954,0.00070214784,0.00007810594,0.00022606006,0.0007042358,0.0001200003],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013768099,0.00093798595,0.0008692113,0.0013792529,0.0004474274,0.00095226936,0.001090735,0.00096348074,0.007119871],"category_scores_gemma":[0.0043638838,0.00025596688,0.000665853,0.0010542239,0.00019542391,0.0029269685,0.0013122695,0.0014954482,0.0035404693],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015715662,0.0009806359,0.0055961,0.0004979068,0.000248783,0.00028684465,0.00032552367,0.020316258,0.05460699,0.009898605,0.044353332,0.8613174],"study_design_scores_gemma":[0.00011678061,0.0004947137,0.0025023578,0.000046003195,0.00021405135,0.00032756774,0.00024661774,0.9375609,0.03966217,0.01192872,0.0068509765,0.00004909856],"about_ca_topic_score_codex":0.0029359413,"about_ca_topic_score_gemma":0.0047521926,"teacher_disagreement_score":0.007119871,"about_ca_system_score_codex":0.00045650516,"about_ca_system_score_gemma":0.0014243943,"threshold_uncertainty_score":0.023818314},"labels":[],"label_agreement":null},{"id":"W3029985558","doi":"","title":"The Nunavut Hansard Inuktitut-English Parallel Corpus 3.0 with Preliminary Machine Translation Results.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Computer science; Sentence; Machine translation; Indigenous; Natural language processing; Indigenous language; Linguistics; Artificial intelligence; Speech recognition","score_opus":0.021495609075024055,"score_gpt":0.266804596939113,"score_spread":0.24530898786408895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029985558","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23886031,0.01079709,0.045088362,0.0033896961,0.0029717367,0.0037742986,0.48888126,0.026990797,0.17924646],"genre_scores_gemma":[0.19331539,0.0015148235,0.08629465,0.0005749202,0.00026042687,0.0028971455,0.6639236,0.006602052,0.04461699],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9966558,0.0015751676,0.00026034165,0.0006720349,0.00054772926,0.00028887024],"domain_scores_gemma":[0.995404,0.0012170307,0.00013294669,0.00076404336,0.0020804713,0.0004015454],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032643853,0.0014922242,0.0013098557,0.004440198,0.0039448505,0.0026185962,0.0018982975,0.0010983951,0.04886709],"category_scores_gemma":[0.010710484,0.0007707741,0.0005438368,0.0047834236,0.0009296555,0.002248779,0.0040594917,0.0013280001,0.027044136],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0044780304,0.0011471915,0.0065202015,0.0051946435,0.00041680204,0.0024659205,0.005251723,0.004453098,0.035370547,0.014673732,0.6433319,0.2766962],"study_design_scores_gemma":[0.0018201751,0.00060397125,0.03017794,0.0010080598,0.0005303608,0.0022773647,0.0041374443,0.012494155,0.045275025,0.0064273127,0.89499134,0.0002567512],"about_ca_topic_score_codex":0.072437026,"about_ca_topic_score_gemma":0.0986783,"teacher_disagreement_score":0.072437026,"about_ca_system_score_codex":0.0017997066,"about_ca_system_score_gemma":0.0063329083,"threshold_uncertainty_score":0.16347665},"labels":[],"label_agreement":null},{"id":"W3029996834","doi":"","title":"Corpus of Chinese Dynastic Histories: Gender Analysis over Two Millennia.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Chinese history and philosophy","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Lexicon; Computer science; Dominance (genetics); License; Focus (optics); Semantics (computer science); Linguistics; Natural language processing; History; Classical Chinese; Space (punctuation); Corpus linguistics; Artificial intelligence","score_opus":0.03543237694785305,"score_gpt":0.3434333187503363,"score_spread":0.3080009418024832,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3029996834","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.47861803,0.007927224,0.0039621494,0.0022374492,0.00046884036,0.00037677464,0.41755828,0.0003756185,0.08847569],"genre_scores_gemma":[0.7250482,0.0030320291,0.0034771136,0.00021394771,0.00020027372,0.000973646,0.23655939,0.00023809461,0.03025736],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9995894,0.0000840613,0.000058098703,0.00008718431,0.000120145676,0.00006108245],"domain_scores_gemma":[0.9971975,0.0010185472,0.00029709408,0.0003099107,0.0008871655,0.0002897708],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010479685,0.00040370642,0.00031724616,0.0073420275,0.001881488,0.00083968433,0.0006932876,0.00030104915,0.010987295],"category_scores_gemma":[0.0051070987,0.0001640052,0.00015515626,0.014448511,0.0007114673,0.0011202258,0.0016840076,0.0004161498,0.001835542],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007947475,0.00014775494,0.14013065,0.0048058387,0.00014769535,0.0030212214,0.082012646,0.00062372646,0.00935282,0.031986896,0.3378414,0.38913456],"study_design_scores_gemma":[0.00002813734,0.000034289,0.49699268,0.00039818164,0.000102186634,0.0005510287,0.012125746,0.0006628698,0.0024907535,0.0014658255,0.48510793,0.00004044893],"about_ca_topic_score_codex":0.1148642,"about_ca_topic_score_gemma":0.16203529,"teacher_disagreement_score":0.1148642,"about_ca_system_score_codex":0.0024080032,"about_ca_system_score_gemma":0.006122392,"threshold_uncertainty_score":0.22839123},"labels":[],"label_agreement":null},{"id":"W3030000665","doi":"","title":"FAB: The French Absolute Beginner Corpus for Pronunciation Training.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Phonetics and Phonology Research","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Pronunciation; Context (archaeology); Natural language processing; Set (abstract data type); Partition (number theory); Artificial intelligence; Training set; Speech recognition; Word (group theory); Linguistics; Mathematics","score_opus":0.10830507150931382,"score_gpt":0.37666659247843487,"score_spread":0.268361520969121,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3030000665","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.058683723,0.003757578,0.014370768,0.0011862575,0.0007280223,0.0005631123,0.79904073,0.017515358,0.10415443],"genre_scores_gemma":[0.1647836,0.00086948805,0.019596815,0.0004153172,0.00024306001,0.0011335921,0.7637794,0.0045057717,0.04467303],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9986871,0.00044190066,0.000108384935,0.00028391558,0.00034405626,0.00013468119],"domain_scores_gemma":[0.9966312,0.00096505496,0.00010265117,0.0005223898,0.0014458715,0.0003328252],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013633812,0.0015232197,0.0007580734,0.003618473,0.0013788299,0.0016040305,0.0011813004,0.001078228,0.083042465],"category_scores_gemma":[0.0051847226,0.00029432873,0.00029397564,0.001912077,0.00061461044,0.0010040236,0.0017260918,0.0010433447,0.04305807],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009923768,0.0002524861,0.0049103876,0.0015001274,0.000059482223,0.00067977654,0.0020002355,0.0010168749,0.014560591,0.005670175,0.7009569,0.26740065],"study_design_scores_gemma":[0.0002712409,0.00017091018,0.049842723,0.00038527325,0.000045765617,0.0014115239,0.0013865334,0.0015112855,0.0107994815,0.0019538947,0.93212587,0.00009553322],"about_ca_topic_score_codex":0.060453814,"about_ca_topic_score_gemma":0.055252105,"teacher_disagreement_score":0.083042465,"about_ca_system_score_codex":0.0012269499,"about_ca_system_score_gemma":0.002657938,"threshold_uncertainty_score":0.27780473},"labels":[],"label_agreement":null},{"id":"W3030151190","doi":"","title":"Evaluating Approaches to Personalizing Language Models","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Perplexity; Computer science; Language model; Personalization; Artificial intelligence; Natural language processing; Adaptation (eye); Word (group theory); World Wide Web","score_opus":0.3079769636386057,"score_gpt":0.34640647007214864,"score_spread":0.038429506433542926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3030151190","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.47249722,0.011102872,0.4950416,0.002418332,0.0003952516,0.0018100537,0.0012021321,0.0050395085,0.010493055],"genre_scores_gemma":[0.8015619,0.0016179744,0.19186635,0.00035119406,0.00017129803,0.0006186442,0.0016719935,0.00032032977,0.0018202517],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.96263033,0.02760296,0.0018488351,0.00280948,0.004513926,0.0005944825],"domain_scores_gemma":[0.7904698,0.19104068,0.002701092,0.008089342,0.0058782957,0.0018207245],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04093151,0.002124774,0.0016757558,0.004737976,0.0013210456,0.0045657884,0.002397554,0.003610964,0.0020054919],"category_scores_gemma":[0.127417,0.0009409156,0.0016472097,0.0028031,0.0014662419,0.008082716,0.0033375607,0.003214137,0.0006081397],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0063347244,0.0036556716,0.031961188,0.001922498,0.0026349765,0.00016148663,0.0029271962,0.23214753,0.0059536565,0.013220659,0.007822404,0.691258],"study_design_scores_gemma":[0.0006525703,0.0014104656,0.004714962,0.00019289408,0.0014684085,0.00011963777,0.0008012542,0.95469844,0.007566678,0.025668152,0.0025953406,0.00011107546],"about_ca_topic_score_codex":0.00896824,"about_ca_topic_score_gemma":0.012189786,"teacher_disagreement_score":0.04093151,"about_ca_system_score_codex":0.0034036443,"about_ca_system_score_gemma":0.0029886658,"threshold_uncertainty_score":0.21646911},"labels":[],"label_agreement":null},{"id":"W3030267806","doi":"","title":"On The Performance of Time-Pooling Strategies for End-to-End Spoken Language Identification.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"","keywords":"Pooling; Computer science; Benchmark (surveying); Artificial intelligence; Dimension (graph theory); Representation (politics); Set (abstract data type); Spoken language; Identification (biology); Machine learning; Selection (genetic algorithm); Language model; Natural language processing; Test set; Mathematics","score_opus":0.0337936629644121,"score_gpt":0.28381202723240906,"score_spread":0.25001836426799695,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3030267806","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44809765,0.009066168,0.51527435,0.00072534557,0.00071212114,0.00037080713,0.001971234,0.0108594345,0.0129229305],"genre_scores_gemma":[0.81299096,0.000981932,0.16959393,0.00043836256,0.00014890333,0.0002018616,0.0041782865,0.0004376133,0.011028305],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9977842,0.0008316992,0.00018400654,0.00038368747,0.00052809075,0.00028831555],"domain_scores_gemma":[0.9956117,0.0031316748,0.000108556655,0.00028912324,0.0006953469,0.00016356229],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004529137,0.0017912817,0.0012485118,0.0009858542,0.0007086286,0.0015701199,0.0016306954,0.0017966118,0.005151122],"category_scores_gemma":[0.010193691,0.00042177847,0.0005805491,0.00053354405,0.0005035213,0.0029064144,0.0016960875,0.0011592562,0.0024025626],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.008110458,0.0009206165,0.0036248646,0.0006567977,0.0008312168,0.00043860113,0.00025346066,0.09053636,0.07906456,0.0032465558,0.012195159,0.8001214],"study_design_scores_gemma":[0.00014931834,0.0011496574,0.0044236295,0.00004311497,0.0002502191,0.0003913434,0.0002690087,0.91046286,0.07875666,0.0020143776,0.0020238399,0.00006593536],"about_ca_topic_score_codex":0.013417034,"about_ca_topic_score_gemma":0.016990686,"teacher_disagreement_score":0.013417034,"about_ca_system_score_codex":0.00068163837,"about_ca_system_score_gemma":0.001427782,"threshold_uncertainty_score":0.026677907},"labels":[],"label_agreement":null},{"id":"W3030307641","doi":"","title":"An Analysis of Massively Multilingual Neural Machine Translation for Low-Resource Languages.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Machine translation; Natural language processing; Artificial intelligence; Set (abstract data type); Resource (disambiguation); Massively parallel; Programming language","score_opus":0.028529584331923822,"score_gpt":0.33911117094096893,"score_spread":0.3105815866090451,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3030307641","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.812707,0.006692375,0.13668266,0.0024830827,0.00039279927,0.00032965676,0.0062006484,0.008270397,0.026241386],"genre_scores_gemma":[0.94799876,0.00049998175,0.040068656,0.00016683621,0.000095505755,0.00013201268,0.006889428,0.00042215118,0.0037265907],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9980459,0.00097258633,0.0001147329,0.00020766769,0.00049798784,0.00016104638],"domain_scores_gemma":[0.99081296,0.006304179,0.00026977193,0.000719206,0.0017598582,0.0001340282],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032445306,0.0006848348,0.00062654103,0.0014891231,0.00084761705,0.0012757229,0.0010275381,0.0007936074,0.0038095831],"category_scores_gemma":[0.014657015,0.00026044133,0.0004849064,0.0017564453,0.00050133216,0.0020461583,0.00082488917,0.0007674697,0.0011915432],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0047535747,0.001309321,0.022440057,0.0021672866,0.0009268579,0.0020070665,0.00064731494,0.3705645,0.03284,0.03933513,0.055981405,0.46702746],"study_design_scores_gemma":[0.000052623327,0.00024248876,0.006318212,0.00003834103,0.000106382475,0.00024779123,0.00017897367,0.9655091,0.00983749,0.013507703,0.003937971,0.000022932496],"about_ca_topic_score_codex":0.008634785,"about_ca_topic_score_gemma":0.013641089,"teacher_disagreement_score":0.008634785,"about_ca_system_score_codex":0.0012828943,"about_ca_system_score_gemma":0.0012143487,"threshold_uncertainty_score":0.017169058},"labels":[],"label_agreement":null},{"id":"W3030379882","doi":"","title":"On the Creation of a Corpus for Coherence Evaluation of Discursive Units","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Bank of Canada; National Bank of Canada","funders":"","keywords":"Computer science; Coherence (philosophical gambling strategy); Natural language processing; Artificial intelligence; Sentence; Argument (complex analysis); Focus (optics); Classifier (UML); Linguistics; Textual entailment; Logical consequence; Mathematics","score_opus":0.05399198577407965,"score_gpt":0.33059146662545824,"score_spread":0.2765994808513786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3030379882","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20092176,0.0009618101,0.7403103,0.0020966234,0.00040937433,0.0038293994,0.00969297,0.007953777,0.03382402],"genre_scores_gemma":[0.2721308,0.0003204328,0.7033212,0.0002181776,0.00011998366,0.002702215,0.013053351,0.001549172,0.006584631],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9863339,0.008650459,0.0011307396,0.0011134103,0.002434774,0.00033674],"domain_scores_gemma":[0.9444702,0.033751573,0.0014920282,0.005883655,0.013068661,0.0013338375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012335106,0.00076419435,0.000889957,0.0064268326,0.0030776747,0.003905706,0.0022068669,0.0016343232,0.0094846105],"category_scores_gemma":[0.047721807,0.00077330973,0.00046984767,0.0036237217,0.0023958622,0.0069885342,0.005346199,0.0019935889,0.0036392775],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008691405,0.0013138934,0.00950704,0.0014274669,0.000120015924,0.00077853695,0.011883001,0.0114542395,0.0596853,0.07109675,0.05890156,0.772963],"study_design_scores_gemma":[0.0007250828,0.0013874253,0.031472094,0.0010283954,0.00033756468,0.0021642186,0.014517463,0.4315942,0.21971002,0.071592495,0.22490448,0.00056656444],"about_ca_topic_score_codex":0.008010175,"about_ca_topic_score_gemma":0.012790616,"teacher_disagreement_score":0.012335106,"about_ca_system_score_codex":0.0015974978,"about_ca_system_score_gemma":0.003278793,"threshold_uncertainty_score":0.06523508},"labels":[],"label_agreement":null},{"id":"W3030512995","doi":"","title":"Evaluating Sub-word Embeddings in Cross-lingual Models.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Word (group theory); Natural language processing; Artificial intelligence; Lexicon; Task (project management); Space (punctuation); Vocabulary; Resource (disambiguation); Linguistics","score_opus":0.08551581455221315,"score_gpt":0.37443450547402246,"score_spread":0.28891869092180933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3030512995","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.59945846,0.015724069,0.33414683,0.0026594754,0.002871639,0.0007382498,0.012568854,0.013798783,0.018033566],"genre_scores_gemma":[0.87772155,0.0017848946,0.08768651,0.0004930767,0.00041997322,0.00040943627,0.025590025,0.0009744075,0.004920041],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9900303,0.0067518214,0.0006454022,0.0013595822,0.0008760985,0.0003367616],"domain_scores_gemma":[0.9766827,0.017642228,0.00040628115,0.0021908772,0.0025204557,0.00055745756],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0122037,0.0021400321,0.0014876237,0.003390763,0.00099368,0.0033982655,0.001987271,0.002124491,0.0036045534],"category_scores_gemma":[0.034351476,0.00055992434,0.0014978296,0.0028143695,0.0005946276,0.0074437065,0.0040148743,0.0029386184,0.0033044084],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031178433,0.0021573766,0.032403216,0.0013802418,0.0039509004,0.00040197448,0.0011247386,0.12464195,0.0073644007,0.0066092554,0.051671304,0.7651769],"study_design_scores_gemma":[0.0002509017,0.00079652754,0.005848217,0.00019070708,0.0009837814,0.00027100267,0.0011182139,0.9616428,0.007858173,0.013392046,0.0075486708,0.000098973396],"about_ca_topic_score_codex":0.009764874,"about_ca_topic_score_gemma":0.013116697,"teacher_disagreement_score":0.0122037,"about_ca_system_score_codex":0.0010168677,"about_ca_system_score_gemma":0.0018029115,"threshold_uncertainty_score":0.06454009},"labels":[],"label_agreement":null},{"id":"W3030811146","doi":"","title":"Language Modeling with a General Second-Order RNN.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Recurrent neural network; Computer science; Multiplicative function; Treebank; Language model; State space; State (computer science); Artificial intelligence; Space (punctuation); Sequence (biology); Artificial neural network; Algorithm; Mathematics; Statistics; Parsing","score_opus":0.030082924266851083,"score_gpt":0.2736927433707653,"score_spread":0.24360981910391424,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3030811146","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017295765,0.0008975226,0.9679602,0.0005039775,0.0002601156,0.0002161963,0.0015471446,0.0056049284,0.0057142167],"genre_scores_gemma":[0.5485517,0.00071295194,0.42445138,0.0005517228,0.0002582036,0.0007269519,0.0062896786,0.0012298602,0.017227527],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99756193,0.0013589635,0.00011368353,0.00055492076,0.0002461541,0.00016438843],"domain_scores_gemma":[0.99559116,0.003017338,0.00012361129,0.000411178,0.00069477636,0.00016189019],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037478535,0.0013537433,0.00088107376,0.0011506069,0.00054573163,0.0023175825,0.0019909116,0.0018171361,0.007478758],"category_scores_gemma":[0.012037051,0.00059591996,0.0012204493,0.0010524056,0.0004007799,0.0035728288,0.001491677,0.0028192648,0.005345342],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010306039,0.00031895316,0.0025935792,0.00056601997,0.000430009,0.0002837379,0.0004130266,0.36075035,0.013188823,0.021781554,0.017181339,0.58146197],"study_design_scores_gemma":[0.000012908061,0.000039141603,0.00022539415,0.000019988162,0.000030603933,0.000042728243,0.000029395691,0.9884133,0.0023528757,0.0071790935,0.0016424201,0.0000122215615],"about_ca_topic_score_codex":0.013356977,"about_ca_topic_score_gemma":0.017959462,"teacher_disagreement_score":0.013356977,"about_ca_system_score_codex":0.0011226962,"about_ca_system_score_gemma":0.001774111,"threshold_uncertainty_score":0.026558459},"labels":[],"label_agreement":null},{"id":"W3031043157","doi":"","title":"WEXEA: Wikipedia EXhaustive Entity Annotation","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Annotation; Hyperlink; Information retrieval; Entity linking; Named-entity recognition; Relationship extraction; Natural language processing; Task (project management); Publication; Information extraction; Named entity; Relation (database); Artificial intelligence; World Wide Web; Web page; Knowledge base; Database","score_opus":0.033941972694160195,"score_gpt":0.2878343625034358,"score_spread":0.2538923898092756,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3031043157","genre_codex":"dataset","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12378736,0.006735745,0.13695626,0.0022883858,0.003138864,0.0057238187,0.47861922,0.198221,0.04452935],"genre_scores_gemma":[0.113951795,0.0010466365,0.13273951,0.0007503279,0.00023693552,0.0031164684,0.7258298,0.0084160585,0.013912545],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98602146,0.005578929,0.0016300832,0.0026508265,0.0032578853,0.0008607636],"domain_scores_gemma":[0.96798193,0.014577806,0.00095654605,0.007211316,0.0073012803,0.0019710786],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009988223,0.0036354358,0.002159588,0.01348401,0.0038636462,0.0045115845,0.003977229,0.0029544048,0.025079422],"category_scores_gemma":[0.035603423,0.0012709717,0.0019265752,0.0069316532,0.0011989367,0.00945017,0.008674054,0.0028449611,0.017614784],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0026461824,0.0018446546,0.010073547,0.005221295,0.0013306037,0.00085056655,0.0013079469,0.0065058754,0.01190307,0.0051574344,0.6825647,0.27059412],"study_design_scores_gemma":[0.002505266,0.0021884916,0.034679085,0.0022721593,0.0022424713,0.0028547859,0.004811017,0.18303807,0.079982124,0.022038521,0.6624891,0.00089895906],"about_ca_topic_score_codex":0.03309637,"about_ca_topic_score_gemma":0.043944776,"teacher_disagreement_score":0.03309637,"about_ca_system_score_codex":0.0014597713,"about_ca_system_score_gemma":0.0052247955,"threshold_uncertainty_score":0.08389902},"labels":[],"label_agreement":null},{"id":"W3031229455","doi":"","title":"SEDAR: a Large Scale French-English Financial Domain Parallel Corpus","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Machine translation; Computer science; Domain (mathematical analysis); Preprocessor; Natural language processing; Sentence; Translation (biology); Artificial intelligence; Scale (ratio); Speech recognition; Chemistry","score_opus":0.011982349944706733,"score_gpt":0.26276082289982794,"score_spread":0.25077847295512123,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3031229455","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35992762,0.004966531,0.053512886,0.0040016077,0.0013086969,0.0013637644,0.5016189,0.019606413,0.05369358],"genre_scores_gemma":[0.27266982,0.0011434921,0.061005443,0.0008267665,0.0003155982,0.0012218576,0.6451708,0.0020670972,0.015579072],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9985623,0.00055246404,0.00013152753,0.00033256054,0.0002925288,0.0001287136],"domain_scores_gemma":[0.99505156,0.0021218685,0.00017363389,0.0005859228,0.0017391042,0.000327863],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017739303,0.0012410708,0.0007534519,0.00438736,0.0021277901,0.0016771251,0.0014399004,0.0014330586,0.022219501],"category_scores_gemma":[0.0070475866,0.0004436807,0.0006326935,0.0031112318,0.001014873,0.001985593,0.0018569325,0.0013230087,0.009744931],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024387655,0.0015369855,0.013759375,0.0045031663,0.00059857377,0.00497691,0.0031154943,0.008787898,0.05996235,0.021161687,0.60642755,0.27273127],"study_design_scores_gemma":[0.0012913563,0.0005878602,0.070052415,0.00045189916,0.00053775986,0.0044065216,0.00410619,0.03391825,0.033356875,0.008594587,0.84239507,0.00030119286],"about_ca_topic_score_codex":0.04281013,"about_ca_topic_score_gemma":0.044907823,"teacher_disagreement_score":0.04281013,"about_ca_system_score_codex":0.0014551517,"about_ca_system_score_gemma":0.0034204018,"threshold_uncertainty_score":0.08512193},"labels":[],"label_agreement":null},{"id":"W3031309641","doi":"","title":"Temporal Histories of Epidemic Events (THEE): A Case Study in Temporal Annotation for Public Health","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Public Health Agency of Canada; University of Toronto","funders":"","keywords":"Annotation; Computer science; Metadata; Domain (mathematical analysis); Event (particle physics); Public domain; Information retrieval; Process (computing); Temporal annotation; Style (visual arts); Natural language processing; Artificial intelligence; World Wide Web; History; Natural language","score_opus":0.10611195314227607,"score_gpt":0.38414537583421693,"score_spread":0.27803342269194087,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3031309641","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45811802,0.004313724,0.35909176,0.023391493,0.0005723654,0.0018035175,0.08893823,0.005998227,0.057772692],"genre_scores_gemma":[0.69975775,0.0017769837,0.25279933,0.001084466,0.00013041451,0.0006450691,0.034447096,0.0008909113,0.008468001],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9950819,0.0023980301,0.00069786457,0.0005816004,0.0010723592,0.00016823814],"domain_scores_gemma":[0.9345662,0.05384064,0.0034624748,0.002972795,0.004094421,0.0010635541],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011208732,0.00041004037,0.00031335675,0.0040950803,0.001968867,0.0022609774,0.0011847443,0.0012828966,0.0037511042],"category_scores_gemma":[0.035589248,0.00025812525,0.0005320865,0.006504504,0.0013163966,0.0053196033,0.0024115993,0.0012959995,0.0006549497],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019250576,0.0007954218,0.15611261,0.0064433375,0.00026961314,0.014136548,0.054091778,0.023577573,0.015136072,0.21556996,0.08882712,0.4231149],"study_design_scores_gemma":[0.0001421909,0.0002914542,0.07159671,0.0020517183,0.00036825455,0.007154112,0.03912747,0.12531523,0.022558775,0.09470068,0.6364335,0.00025989904],"about_ca_topic_score_codex":0.038677078,"about_ca_topic_score_gemma":0.046086323,"teacher_disagreement_score":0.038677078,"about_ca_system_score_codex":0.002408865,"about_ca_system_score_gemma":0.004569238,"threshold_uncertainty_score":0.07690388},"labels":[],"label_agreement":null},{"id":"W3032025286","doi":"","title":"Cifu: a Frequency Lexicon of Hong Kong Cantonese.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Linguistic Variation and Morphology","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Lexicon; Computer science; Lexical database; Word lists by frequency; Natural language processing; Lexical diversity; Artificial intelligence; Linguistics; Word (group theory); Phonology; Speech recognition; Vocabulary","score_opus":0.0484546324202565,"score_gpt":0.34967562486616527,"score_spread":0.3012209924459088,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032025286","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5587788,0.0040322663,0.040176887,0.000728969,0.00036602435,0.0011857894,0.2705348,0.005719889,0.118476585],"genre_scores_gemma":[0.8344222,0.0010261936,0.028549341,0.00011553578,0.000048211146,0.0008829161,0.115254514,0.00069689454,0.01900418],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9996526,0.00007433435,0.000095318755,0.00006451681,0.00007331763,0.00003993783],"domain_scores_gemma":[0.99833363,0.0003899707,0.00011529342,0.00017313502,0.0008681649,0.00011976094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007203024,0.0010296063,0.00047683163,0.005183459,0.0013286569,0.0016851673,0.0007218004,0.0002851599,0.013795663],"category_scores_gemma":[0.0027572396,0.00027907416,0.00022577877,0.0053672963,0.000497779,0.0015175511,0.0009699404,0.00038443043,0.0021564942],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001007124,0.00020761405,0.1042228,0.0042307246,0.00030741448,0.0028207882,0.034212556,0.0036250306,0.03339489,0.036299165,0.24344267,0.5362292],"study_design_scores_gemma":[0.000117848365,0.00018792662,0.3539337,0.00064575847,0.00040672327,0.0024197444,0.013810113,0.009488307,0.008754553,0.00346575,0.606543,0.00022657082],"about_ca_topic_score_codex":0.2457548,"about_ca_topic_score_gemma":0.22039442,"teacher_disagreement_score":0.2457548,"about_ca_system_score_codex":0.0026848246,"about_ca_system_score_gemma":0.0048229448,"threshold_uncertainty_score":0.48864865},"labels":[],"label_agreement":null},{"id":"W3032097118","doi":"","title":"Multilingual Corpus Creation for Multilingual Semantic Similarity Task","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo; Western University","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Task (project management); Semantic similarity; Similarity (geometry); Sentence; Focus (optics); Information retrieval","score_opus":0.04671167661381604,"score_gpt":0.32804333242450934,"score_spread":0.2813316558106933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032097118","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35983387,0.003532889,0.4620368,0.0016624393,0.0028510084,0.0042803804,0.06052716,0.04745438,0.05782105],"genre_scores_gemma":[0.4802123,0.00089642114,0.35861674,0.0004036339,0.00046344512,0.004078203,0.13491991,0.0045257583,0.015883502],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99586403,0.001499945,0.0005469307,0.0010226321,0.0007505556,0.00031594414],"domain_scores_gemma":[0.9937643,0.0023132055,0.00016183773,0.0010248956,0.0022586256,0.00047712232],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034991994,0.0018633584,0.0014283892,0.0063913604,0.003177388,0.0030685237,0.0016775157,0.0015652209,0.021942575],"category_scores_gemma":[0.011471543,0.0006827826,0.0014150613,0.0038132095,0.000722763,0.0056813043,0.0056137377,0.0023508451,0.0104427505],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023537402,0.0016432743,0.009811264,0.003140762,0.000514995,0.0018910714,0.0043509165,0.0079651065,0.078197636,0.02039156,0.183147,0.6865927],"study_design_scores_gemma":[0.0011729263,0.0015919978,0.024218384,0.00068477704,0.0014881822,0.0050814375,0.012245944,0.32118014,0.23382759,0.031117897,0.36673254,0.00065813307],"about_ca_topic_score_codex":0.008061717,"about_ca_topic_score_gemma":0.008850508,"teacher_disagreement_score":0.021942575,"about_ca_system_score_codex":0.0011205545,"about_ca_system_score_gemma":0.0038726877,"threshold_uncertainty_score":0.073405266},"labels":[],"label_agreement":null},{"id":"W3032201847","doi":"","title":"SpiCE: A New Open-Access Corpus of Conversational Bilingual Speech in Cantonese and English.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Transcription (linguistics); Spice; Sentence; Annotation; Natural language processing; Speech corpus; Storyboard; Phonetic transcription; Speech recognition; Linguistics; Artificial intelligence; Speech synthesis; Multimedia; Engineering","score_opus":0.04815606913529093,"score_gpt":0.3652228133420278,"score_spread":0.31706674420673686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032201847","genre_codex":"dataset","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":"dataset","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.40316015,0.0030706744,0.015206153,0.0011278667,0.0005689238,0.0016713955,0.53317827,0.0050600506,0.036956467],"genre_scores_gemma":[0.33814454,0.0006790278,0.01744675,0.00025583402,0.00015976129,0.0026527701,0.62764704,0.00082365406,0.012190576],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99856573,0.00044234341,0.00018234558,0.00032609949,0.00032100308,0.00016244932],"domain_scores_gemma":[0.9952591,0.0015402322,0.0002556547,0.0006179137,0.0017743856,0.00055265776],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.0014105969,0.0011014863,0.0007409334,0.003325559,0.0017928317,0.0014021785,0.0013202407,0.00087692705,0.01745474],"category_scores_gemma":[0.006742451,0.00034594527,0.00029785704,0.0027798554,0.0008624537,0.0016210782,0.0029110142,0.0010025463,0.004922333],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003967754,0.0014017834,0.04658138,0.007127294,0.00040796265,0.0042429096,0.019991465,0.00279955,0.12522815,0.008758069,0.43091056,0.34858322],"study_design_scores_gemma":[0.00090133515,0.0005803974,0.36788765,0.0011038587,0.00040405127,0.0036140655,0.01548167,0.009502767,0.027733969,0.0031942884,0.5692571,0.00033883445],"about_ca_topic_score_codex":0.06782387,"about_ca_topic_score_gemma":0.102579564,"teacher_disagreement_score":0.99867976,"about_ca_system_score_codex":0.0011679083,"about_ca_system_score_gemma":0.003678437,"threshold_uncertainty_score":0.13485819},"labels":[],"label_agreement":null},{"id":"W3032216439","doi":"","title":"Fine-grained Morphosyntactic Analysis and Generation Tools for More Than One Thousand Languages.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Natural language processing; Security token; Metric (unit); Artificial intelligence; Machine translation; Parallel corpora; Linguistics; Engineering","score_opus":0.04463166772679519,"score_gpt":0.3218636676018755,"score_spread":0.2772319998750803,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032216439","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18066983,0.00363957,0.55680275,0.0012183308,0.0006891248,0.0014023639,0.044170093,0.18337521,0.028032757],"genre_scores_gemma":[0.3336333,0.0007437355,0.54982096,0.00040786783,0.00009308986,0.0010968086,0.09082494,0.009916182,0.013463074],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9967332,0.0010008742,0.00040569223,0.0006314017,0.0010218875,0.00020709017],"domain_scores_gemma":[0.9929865,0.0029451572,0.00024434744,0.002053443,0.00144296,0.00032762656],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039643203,0.001301012,0.00075440604,0.002481219,0.0008222492,0.0017491116,0.001954132,0.0010906454,0.013137224],"category_scores_gemma":[0.010400278,0.0007572334,0.001064951,0.0022704222,0.00069203135,0.004430672,0.0030985773,0.0013901739,0.0068687205],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015034997,0.0007974874,0.0074149254,0.001710637,0.00043932433,0.00067120406,0.0016557008,0.011785039,0.07274408,0.018870786,0.11556646,0.7668409],"study_design_scores_gemma":[0.0012694267,0.0013775029,0.023802804,0.0006670393,0.00071872596,0.0016302576,0.0021072216,0.24682459,0.2704061,0.06490397,0.38587004,0.00042233954],"about_ca_topic_score_codex":0.005240927,"about_ca_topic_score_gemma":0.006514921,"teacher_disagreement_score":0.013137224,"about_ca_system_score_codex":0.00090172986,"about_ca_system_score_gemma":0.0018542582,"threshold_uncertainty_score":0.043948412},"labels":[],"label_agreement":null},{"id":"W3032420555","doi":"","title":"Cooking Up a Neural-based Model for Recipe Classification","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Bank of Canada; Université du Québec à Montréal; Polytechnique Montréal; National Bank of Canada","funders":"","keywords":"Embedding; Computer science; Task (project management); Artificial intelligence; Layer (electronics); Macro; Artificial neural network; Recipe; Natural language processing; Language model; Deep learning; State (computer science); Machine learning; Pattern recognition (psychology); Algorithm; Engineering","score_opus":0.1262190574827426,"score_gpt":0.33255425756928014,"score_spread":0.20633520008653755,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032420555","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12604907,0.0014224768,0.85286915,0.002731309,0.00054282474,0.00029316128,0.0014398124,0.00626941,0.0083828],"genre_scores_gemma":[0.8122604,0.000601353,0.1665064,0.0006877583,0.0002814848,0.00028451695,0.0024697217,0.0003468558,0.016561573],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99935263,0.00020777089,0.00004341989,0.00021234174,0.00010483576,0.000079018646],"domain_scores_gemma":[0.9986155,0.00069225486,0.000041972424,0.00015172083,0.0004315446,0.000066983266],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019033289,0.0009264647,0.00083719013,0.0010558597,0.0006376417,0.0017454786,0.0014891529,0.001941561,0.005485358],"category_scores_gemma":[0.004249679,0.00055980054,0.0011008794,0.00088414794,0.000401279,0.0028936656,0.00094042806,0.0028414435,0.0026187405],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00092731393,0.0007059289,0.005541811,0.00019267191,0.00044041252,0.00012783178,0.00016874631,0.3341322,0.014720538,0.009172742,0.016085638,0.61778414],"study_design_scores_gemma":[0.000011092352,0.00003962978,0.00034505484,0.000009184121,0.000037286223,0.000016459197,0.000013457865,0.9937894,0.0023219467,0.0028702596,0.00053505506,0.000011150282],"about_ca_topic_score_codex":0.020496408,"about_ca_topic_score_gemma":0.026765937,"teacher_disagreement_score":0.020496408,"about_ca_system_score_codex":0.0014638801,"about_ca_system_score_gemma":0.0011444511,"threshold_uncertainty_score":0.0407542},"labels":[],"label_agreement":null},{"id":"W3032491810","doi":"","title":"GM-RKB WikiText Error Correction Task and Baselines.","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Computer science; Error detection and correction; Artificial intelligence; Natural language processing; Task (project management); Language model; Ground truth; Speech recognition; Information retrieval; Machine learning; Algorithm","score_opus":0.021723023275288137,"score_gpt":0.2949398695453348,"score_spread":0.27321684627004666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032491810","genre_codex":"dataset","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.28748575,0.0072319712,0.08066499,0.0037463824,0.007037636,0.006965436,0.29613188,0.212249,0.09848691],"genre_scores_gemma":[0.25482303,0.0008243613,0.11357129,0.0021436512,0.0005430844,0.006159883,0.56284237,0.0129292095,0.046163145],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9870745,0.004811199,0.0017039669,0.002943043,0.0027793534,0.000688035],"domain_scores_gemma":[0.96526104,0.013067847,0.0014044415,0.010214271,0.008017174,0.002035249],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010597548,0.0035350733,0.0019420682,0.0033187147,0.0021810762,0.0036866704,0.0046061273,0.004208244,0.02160893],"category_scores_gemma":[0.053720277,0.0009634225,0.001313136,0.0027590797,0.001208548,0.005665122,0.0072748796,0.00477811,0.033268522],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0052235345,0.0038126973,0.0076753683,0.004758002,0.0007832855,0.000609575,0.0009531707,0.0057441215,0.025251532,0.0020099734,0.6445594,0.2986194],"study_design_scores_gemma":[0.007347206,0.004299547,0.09471348,0.0019245917,0.0019675263,0.004745079,0.003159251,0.1546687,0.16746335,0.013714513,0.5449161,0.001080669],"about_ca_topic_score_codex":0.013923359,"about_ca_topic_score_gemma":0.017720465,"teacher_disagreement_score":0.02160893,"about_ca_system_score_codex":0.0011118863,"about_ca_system_score_gemma":0.0034115305,"threshold_uncertainty_score":0.07228905},"labels":[],"label_agreement":null},{"id":"W3032504444","doi":"","title":"PhonBank and Data Sharing: Recent Developments in European Portuguese","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Semantic Web and Ontologies","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Portuguese; Computer science; Data sharing; Data science","score_opus":0.12966803402925817,"score_gpt":0.3399750691332857,"score_spread":0.2103070351040275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3032504444","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33730525,0.04050773,0.31568593,0.032345742,0.002291256,0.0017633416,0.05358126,0.013898629,0.20262095],"genre_scores_gemma":[0.63975495,0.01731842,0.22594802,0.002528122,0.00082177174,0.0012169869,0.08412026,0.005927137,0.022364337],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9744207,0.010253475,0.0031890247,0.0040387856,0.007047218,0.0010507241],"domain_scores_gemma":[0.9431267,0.02833678,0.0040474725,0.01155917,0.010362463,0.0025674528],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.046494503,0.00095110683,0.0014100012,0.0112708025,0.002868735,0.010721847,0.0036663795,0.0023057603,0.007652842],"category_scores_gemma":[0.061636083,0.0006788203,0.00083254307,0.022228919,0.004384232,0.01909598,0.0076539824,0.0017270212,0.0020270552],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018287019,0.00070662407,0.027897835,0.004440672,0.00035413023,0.00075133773,0.01117286,0.005210933,0.007031627,0.16793352,0.046196282,0.7264754],"study_design_scores_gemma":[0.0002753717,0.00036001284,0.059779733,0.004016492,0.0006083522,0.0013384126,0.013071622,0.02074573,0.020146083,0.084004186,0.79531866,0.00033525698],"about_ca_topic_score_codex":0.021392703,"about_ca_topic_score_gemma":0.012409664,"teacher_disagreement_score":0.9963336,"about_ca_system_score_codex":0.005165192,"about_ca_system_score_gemma":0.012260617,"threshold_uncertainty_score":0.24588937},"labels":[],"label_agreement":null},{"id":"W3122399969","doi":"10.1007/s10579-020-09519-z","title":"Exploring the role of lexis and grammar for the stable identification of register in an unrestricted corpus of web documents","year":2021,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; Brockhouse Institute for Materials Research; Brock University","funders":"National Science Foundation of Sri Lanka; Social Sciences and Humanities Research Council of Canada; Turun Yliopisto; Emil Aaltosen Säätiö; AGE-WELL; National Science Foundation","keywords":"Lexis; Register (sociolinguistics); Computer science; Natural language processing; Variation (astronomy); Artificial intelligence; Corpus linguistics; Identification (biology); Linguistics; Text corpus","score_opus":0.05024759076246877,"score_gpt":0.3165752133910301,"score_spread":0.26632762262856136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3122399969","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.74908245,0.002278871,0.22822389,0.005804178,0.00013401487,0.00021147594,0.001898811,0.0013280645,0.01103816],"genre_scores_gemma":[0.9544843,0.0004414921,0.040895242,0.00020979805,0.00006928164,0.0001268929,0.0021465234,0.000320802,0.0013057092],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9966619,0.002075292,0.0001680064,0.0007769508,0.0002063381,0.0001114537],"domain_scores_gemma":[0.96832216,0.026222484,0.0014701281,0.0024877246,0.0011075392,0.00038987646],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0093822945,0.0008739304,0.0008480606,0.0035224175,0.001760971,0.005902129,0.0013116822,0.0014121163,0.0024349797],"category_scores_gemma":[0.045744207,0.0007533643,0.0013115305,0.0021675392,0.0032064586,0.009930177,0.002618184,0.0036517715,0.0011714369],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001468755,0.000623674,0.24601637,0.0008695507,0.00096391677,0.0013044752,0.014271765,0.21931721,0.015526873,0.11786512,0.011755605,0.37001666],"study_design_scores_gemma":[0.000041722968,0.000088275396,0.016617645,0.00011382177,0.00012043375,0.00020283823,0.0013933112,0.8837363,0.002129802,0.09122659,0.004254509,0.00007474152],"about_ca_topic_score_codex":0.013598489,"about_ca_topic_score_gemma":0.020875715,"teacher_disagreement_score":0.013598489,"about_ca_system_score_codex":0.0019623954,"about_ca_system_score_gemma":0.002071826,"threshold_uncertainty_score":0.0496189},"labels":[],"label_agreement":null},{"id":"W4285793742","doi":"10.1007/s10579-022-09602-7","title":"Speech acts in the Dutch COVID-19 Press Conferences","year":2022,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Text Readability and Simplification","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek; Canadian Institute of Steel Construction","keywords":"Speech act; Computer science; Metadata; Linguistics; Reciprocal; Coronavirus disease 2019 (COVID-19); Direct speech; Natural language processing; Mean reciprocal rank; Artificial intelligence; Classifier (UML); Transformer; British National Corpus; World Wide Web; Medicine; Philosophy","score_opus":0.07035566898821673,"score_gpt":0.3433580144822387,"score_spread":0.27300234549402197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285793742","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.67538005,0.004962876,0.009906097,0.0007445315,0.00047722168,0.0011555182,0.26917276,0.0006884756,0.03751247],"genre_scores_gemma":[0.5628696,0.001480236,0.012971321,0.00019153976,0.00017672474,0.0029832458,0.40424028,0.00051687117,0.014570237],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9961696,0.0015404914,0.00048451318,0.0005978607,0.0010128904,0.00019464862],"domain_scores_gemma":[0.9933873,0.0036737,0.0007491782,0.0003809411,0.0015013178,0.00030762766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019154358,0.0007202451,0.00049166655,0.0027781164,0.0008458489,0.0016820848,0.0006636202,0.0006525142,0.009995425],"category_scores_gemma":[0.008203429,0.0003265139,0.00033282157,0.0033805512,0.0007331179,0.0011620932,0.0014632388,0.00069182104,0.0040895706],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029758094,0.0011440421,0.07848129,0.016081758,0.0002850629,0.006679327,0.06905492,0.007103841,0.05590142,0.009845426,0.35263732,0.39980975],"study_design_scores_gemma":[0.00028420961,0.000253843,0.402066,0.0015664167,0.00013953146,0.0024216683,0.019940639,0.010560202,0.016503513,0.0015049687,0.54450357,0.00025552913],"about_ca_topic_score_codex":0.024234831,"about_ca_topic_score_gemma":0.026854454,"teacher_disagreement_score":0.024234831,"about_ca_system_score_codex":0.0018600427,"about_ca_system_score_gemma":0.0012008534,"threshold_uncertainty_score":0.048187554},"labels":[],"label_agreement":null},{"id":"W4393949820","doi":"10.1007/s10579-024-09720-4","title":"Depression symptoms modelling from social media text: an LLM driven semi-supervised learning approach","year":2024,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Mental Health via Writing","field":"Psychology","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs; Alberta Machine Intelligence Institute","keywords":"Social media; Depression (economics); Psychology; Artificial intelligence; Computer science; Cognitive psychology; Natural language processing; World Wide Web","score_opus":0.053996650337715724,"score_gpt":0.36384960739755323,"score_spread":0.3098529570598375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4393949820","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.094718814,0.0005388293,0.89710826,0.0009710612,0.00009026416,0.00031642956,0.001020205,0.0041278386,0.0011082017],"genre_scores_gemma":[0.7661081,0.00012515073,0.22650611,0.0006267092,0.00017349908,0.0004379642,0.002842323,0.00022279572,0.002957289],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99827206,0.00073510193,0.00014591133,0.0004739781,0.00024415375,0.00012874727],"domain_scores_gemma":[0.9913669,0.0059562814,0.00063122046,0.0006066253,0.001192569,0.00024642676],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033545068,0.00093794527,0.0012328529,0.00110831,0.0004915761,0.0011726499,0.0032981841,0.0017776915,0.0015072289],"category_scores_gemma":[0.008420451,0.0006858988,0.0011613535,0.0007575578,0.0007871129,0.0012725991,0.0016036448,0.0021403101,0.0010395978],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000624489,0.0012734897,0.014042419,0.0004939333,0.00035476324,0.0005556996,0.0006839805,0.52786505,0.0075381184,0.0044220085,0.013458858,0.42868713],"study_design_scores_gemma":[0.00000813917,0.000026711297,0.00026008274,0.000005503316,0.0000063210323,0.000015053417,0.000015926043,0.99713,0.0006462433,0.0016870351,0.00019264239,0.0000063366283],"about_ca_topic_score_codex":0.0050666356,"about_ca_topic_score_gemma":0.007454724,"teacher_disagreement_score":0.0050666356,"about_ca_system_score_codex":0.0010327263,"about_ca_system_score_gemma":0.001151212,"threshold_uncertainty_score":0.017740548},"labels":[],"label_agreement":null},{"id":"W4407752103","doi":"10.1007/s10579-025-09813-8","title":"The narratives of war (NoW) corpus of written testimonies of the Russia-Ukraine war","year":2025,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Mental Health via Writing","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"Social Sciences and Humanities Research Council of Canada; Mitacs; Canada Research Chairs","keywords":"Ukrainian; Documentation; Narrative; Spanish Civil War; History; Psychology; Political science; Linguistics; Law; Computer science","score_opus":0.024401145646535418,"score_gpt":0.369499880025237,"score_spread":0.3450987343787016,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407752103","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84637654,0.011378708,0.007819727,0.002145806,0.00084310526,0.0009887379,0.077187866,0.00016446708,0.05309511],"genre_scores_gemma":[0.9079545,0.0067032883,0.018976707,0.00045223229,0.00019899085,0.0014137875,0.04863334,0.00020846445,0.015458764],"study_design_codex":"qualitative","study_design_gemma":"not_applicable","domain_scores_codex":[0.998616,0.0006049502,0.00028631688,0.00016701114,0.0002536983,0.00007200735],"domain_scores_gemma":[0.99439776,0.003087032,0.00074406475,0.0007308478,0.00087980065,0.0001605347],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014105929,0.00021338096,0.00023619548,0.0032449332,0.0016681809,0.0012291687,0.00040740767,0.0004618004,0.0043725315],"category_scores_gemma":[0.008827239,0.00017418838,0.00011311826,0.0042364607,0.0012406182,0.0011226743,0.0021363187,0.0005371069,0.0006841075],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00063807925,0.00021493083,0.03492993,0.007428119,0.00009362151,0.0065165907,0.47197098,0.0014243876,0.023593169,0.03056093,0.09912701,0.3235022],"study_design_scores_gemma":[0.000016463706,0.000054819968,0.09085731,0.0013052936,0.00004314946,0.002151061,0.08240339,0.00031217252,0.0061994176,0.0014890964,0.8151228,0.000044989425],"about_ca_topic_score_codex":0.0058503044,"about_ca_topic_score_gemma":0.012554423,"teacher_disagreement_score":0.0058503044,"about_ca_system_score_codex":0.001270841,"about_ca_system_score_gemma":0.0018467797,"threshold_uncertainty_score":0.014627576},"labels":[],"label_agreement":null}]}